{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":50160,"databundleVersionId":7921029,"isSourceIdPinned":false}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ── Cell 1: Import libraries ──────────────────────────────\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:11:49.738671Z","iopub.execute_input":"2026-04-19T21:11:49.738890Z","iopub.status.idle":"2026-04-19T21:11:52.860416Z","shell.execute_reply.started":"2026-04-19T21:11:49.738871Z","shell.execute_reply":"2026-04-19T21:11:52.859612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 2: Load data files ───────────────────────────────\n\n# Correct path to training files\npath = '/kaggle/input/competitions/home-credit-credit-risk-model-stability/parquet_files/train/'\n\n# Load base file → has case_id + target (1=default, 0=no default)\ndf_base = pd.read_parquet(path + 'train_base.parquet')\n\n# Load static features → has financial & demographic info per client\ndf_static = pd.concat([\n    pd.read_parquet(path + 'train_static_0_0.parquet'),\n    pd.read_parquet(path + 'train_static_0_1.parquet')\n], ignore_index=True)\n\n# Merge both tables using case_id (unique client ID)\ndf = df_base.merge(df_static, on='case_id', how='left')\n\nprint(f\"✅ Shape: {df.shape}\")\nprint(f\"\\nTarget counts:\\n{df['target'].value_counts()}\")\nprint(f\"\\nDefault rate: {df['target'].mean()*100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:15:29.963655Z","iopub.execute_input":"2026-04-19T21:15:29.963985Z","iopub.status.idle":"2026-04-19T21:15:37.279033Z","shell.execute_reply.started":"2026-04-19T21:15:29.963959Z","shell.execute_reply":"2026-04-19T21:15:37.278284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 3: Quick look at the data ───────────────────────\n\n# Show first 3 rows\nprint(\"=== Sample Data ===\")\ndisplay(df.head(3))\n\n# Show missing values percentage (top 10 columns)\nprint(\"\\n=== Top 10 Missing Values ===\")\nmissing = (df.isnull().sum() / len(df) * 100).sort_values(ascending=False)\nprint(missing.head(10))\n\n# Show column data types\nprint(\"\\n=== Data Types Count ===\")\nprint(df.dtypes.value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:16:01.333872Z","iopub.execute_input":"2026-04-19T21:16:01.334146Z","iopub.status.idle":"2026-04-19T21:16:03.073407Z","shell.execute_reply.started":"2026-04-19T21:16:01.334120Z","shell.execute_reply":"2026-04-19T21:16:03.072808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 4: Drop columns with too many missing values ─────\n\n# Remove columns where more than 40% of data is missing\n# No point trying to fill columns that are almost empty\nmissing_pct = df.isnull().sum() / len(df) * 100\ncols_to_drop = missing_pct[missing_pct > 40].index.tolist()\n\ndf = df.drop(columns=cols_to_drop)\n\nprint(f\"✅ Dropped {len(cols_to_drop)} columns\")\nprint(f\"Remaining shape: {df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:17:33.706983Z","iopub.execute_input":"2026-04-19T21:17:33.707233Z","iopub.status.idle":"2026-04-19T21:17:35.858407Z","shell.execute_reply.started":"2026-04-19T21:17:33.707214Z","shell.execute_reply":"2026-04-19T21:17:35.857821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 5: Separate columns by type ──────────────────────\n\n# Identify columns we don't need for modeling\ncols_to_skip = ['case_id', 'date_decision', 'WEEK_NUM', 'target']\n\n# Numerical columns (float & int) → fill missing with median\nnum_cols = df.select_dtypes(include=['float64', 'int64']).columns.tolist()\nnum_cols = [c for c in num_cols if c not in cols_to_skip]\n\n# Categorical columns (object/text) → fill missing then encode\ncat_cols = df.select_dtypes(include=['object', 'bool']).columns.tolist()\n\nprint(f\"Numerical columns: {len(num_cols)}\")\nprint(f\"Categorical columns: {len(cat_cols)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:17:55.605890Z","iopub.execute_input":"2026-04-19T21:17:55.606135Z","iopub.status.idle":"2026-04-19T21:17:56.493240Z","shell.execute_reply.started":"2026-04-19T21:17:55.606116Z","shell.execute_reply":"2026-04-19T21:17:56.492507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 6: Fill missing values ───────────────────────────\n\n# Fill numerical columns with median (robust to outliers)\ndf[num_cols] = df[num_cols].fillna(df[num_cols].median())\n\n# Fill categorical columns with the most frequent value\nfor col in cat_cols:\n    df[col] = df[col].fillna(df[col].mode()[0])\n\nprint(\"✅ Missing values filled!\")\nprint(f\"Remaining nulls: {df.isnull().sum().sum()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:18:08.731089Z","iopub.execute_input":"2026-04-19T21:18:08.731323Z","iopub.status.idle":"2026-04-19T21:18:14.633532Z","shell.execute_reply.started":"2026-04-19T21:18:08.731302Z","shell.execute_reply":"2026-04-19T21:18:14.632612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 7: Encode categorical columns ────────────────────\n\n# Convert text/category columns to numbers using Label Encoding\n# Neural networks only understand numbers, not text\nfrom sklearn.preprocessing import LabelEncoder\n\nle = LabelEncoder()\nfor col in cat_cols:\n    df[col] = le.fit_transform(df[col].astype(str))\n\nprint(\"✅ Categorical columns encoded!\")\nprint(f\"Final shape: {df.shape}\")\nprint(f\"\\nData types now:\\n{df.dtypes.value_counts()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:18:52.356068Z","iopub.execute_input":"2026-04-19T21:18:52.356322Z","iopub.status.idle":"2026-04-19T21:18:56.525968Z","shell.execute_reply.started":"2026-04-19T21:18:52.356300Z","shell.execute_reply":"2026-04-19T21:18:56.525335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Check available columns for feature engineering ────────\nprint(df.columns.tolist())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:21:11.497984Z","iopub.execute_input":"2026-04-19T21:21:11.498269Z","iopub.status.idle":"2026-04-19T21:21:11.502668Z","shell.execute_reply.started":"2026-04-19T21:21:11.498247Z","shell.execute_reply":"2026-04-19T21:21:11.502034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 8: Feature Engineering ───────────────────────────\n\n# Ratio of annuity to total debt (how much they pay vs what they owe)\ndf['annuity_to_debt'] = df['annuity_780A'] / (df['totaldebt_9A'] + 1)\n\n# Ratio of total settled to total debt (repayment progress)\ndf['settled_to_debt'] = df['totalsettled_863A'] / (df['totaldebt_9A'] + 1)\n\n# Ratio of credit amount to main income (affordability)\ndf['credit_to_income'] = df['credamount_770A'] / (df['maininc_215A'] + 1)\n\n# Ratio of current debt to credit amount (utilization)\ndf['debt_to_credit'] = df['currdebt_22A'] / (df['credamount_770A'] + 1)\n\n# Number of late payments ratio\ndf['late_pay_ratio'] = df['numinstlswithdpd10_728L'] / (df['numinstls_657L'] + 1)\n\nprint(\"✅ New features created!\")\nprint(f\"Shape after engineering: {df.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:21:52.730448Z","iopub.execute_input":"2026-04-19T21:21:52.730937Z","iopub.status.idle":"2026-04-19T21:21:52.759108Z","shell.execute_reply.started":"2026-04-19T21:21:52.730908Z","shell.execute_reply":"2026-04-19T21:21:52.758486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Fix fragmentation warning ──────────────────────────────\n# Defragment the dataframe for better performance\ndf = df.copy()\nprint(\"✅ DataFrame defragmented!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:23:28.877082Z","iopub.execute_input":"2026-04-19T21:23:28.877340Z","iopub.status.idle":"2026-04-19T21:23:29.538353Z","shell.execute_reply.started":"2026-04-19T21:23:28.877318Z","shell.execute_reply":"2026-04-19T21:23:29.537597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 9: Prepare X and y ───────────────────────────────\n\n# X = all features (input to the model)\n# y = target (what we want to predict)\ndrop_cols = ['case_id', 'date_decision', 'WEEK_NUM', 'target']\nX = df.drop(columns=drop_cols)\ny = df['target']\n\nprint(f\"✅ X shape: {X.shape}\")\nprint(f\"✅ y shape: {y.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:23:44.688250Z","iopub.execute_input":"2026-04-19T21:23:44.688497Z","iopub.status.idle":"2026-04-19T21:23:44.927281Z","shell.execute_reply.started":"2026-04-19T21:23:44.688477Z","shell.execute_reply":"2026-04-19T21:23:44.926332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 10: Train / Validation Split ─────────────────────\n\nfrom sklearn.model_selection import train_test_split\n\n# 80% training, 20% validation\n# stratify=y → keeps same default rate in both sets\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n\nprint(f\"✅ Split done!\")\nprint(f\"Train size: {X_train.shape}\")\nprint(f\"Val size:   {X_val.shape}\")\nprint(f\"Train default rate: {y_train.mean()*100:.2f}%\")\nprint(f\"Val default rate:   {y_val.mean()*100:.2f}%\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:23:55.960263Z","iopub.execute_input":"2026-04-19T21:23:55.961130Z","iopub.status.idle":"2026-04-19T21:23:57.749282Z","shell.execute_reply.started":"2026-04-19T21:23:55.961092Z","shell.execute_reply":"2026-04-19T21:23:57.748590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 11: Feature Scaling ───────────────────────────────\n\nfrom sklearn.preprocessing import StandardScaler\n\n# StandardScaler → mean=0, std=1 for all features\n# Neural networks train better when all features on same scale\nscaler = StandardScaler()\n\n# Fit ONLY on train → transform both (avoid data leakage)\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled   = scaler.transform(X_val)\n\nprint(\"✅ Scaling done!\")\nprint(f\"Train mean (should be ~0): {X_train_scaled.mean():.4f}\")\nprint(f\"Train std  (should be ~1): {X_train_scaled.std():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:24:11.050370Z","iopub.execute_input":"2026-04-19T21:24:11.050746Z","iopub.status.idle":"2026-04-19T21:24:12.894844Z","shell.execute_reply.started":"2026-04-19T21:24:11.050727Z","shell.execute_reply":"2026-04-19T21:24:12.894171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 12: Build Neural Network ─────────────────────────\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\n# Set seed for reproducibility\ntf.random.set_seed(42)\n\n# Build the model\nmodel = keras.Sequential([\n    # Input layer\n    layers.Input(shape=(X_train_scaled.shape[1],)),\n    \n    # Hidden layer 1\n    layers.Dense(256, activation='relu'),\n    layers.Dropout(0.3),   # prevent overfitting\n    \n    # Hidden layer 2\n    layers.Dense(128, activation='relu'),\n    layers.Dropout(0.3),\n    \n    # Hidden layer 3\n    layers.Dense(64, activation='relu'),\n    layers.Dropout(0.2),\n    \n    # Output layer → probability between 0 and 1\n    layers.Dense(1, activation='sigmoid')\n])\n\nmodel.summary()\nprint(\"\\n✅ Model created!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:28:03.811996Z","iopub.execute_input":"2026-04-19T21:28:03.812407Z","iopub.status.idle":"2026-04-19T21:28:30.179243Z","shell.execute_reply.started":"2026-04-19T21:28:03.812384Z","shell.execute_reply":"2026-04-19T21:28:30.178741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 13: Compile the model ────────────────────────────\n\n# Adam is the most popular optimizer for neural networks\n# binary_crossentropy is used for yes/no prediction problems\n# AUC measures how well model separates defaults from non-defaults\nmodel.compile(\n    optimizer='adam',\n    loss='binary_crossentropy',\n    metrics=['AUC']\n)\n\nprint(\"✅ Model compiled!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:30:55.411506Z","iopub.execute_input":"2026-04-19T21:30:55.411834Z","iopub.status.idle":"2026-04-19T21:30:55.420109Z","shell.execute_reply.started":"2026-04-19T21:30:55.411809Z","shell.execute_reply":"2026-04-19T21:30:55.419414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 14: Train the model ──────────────────────────────\n\n# early stopping → automatically stops if model stops improving\nearly_stop = keras.callbacks.EarlyStopping(\n    monitor='val_auc',\n    patience=3,\n    restore_best_weights=True,\n    mode='max'\n)\n\n# class_weight → because only 3% of clients defaulted\n# we tell the model \"pay more attention to default cases\"\nhistory = model.fit(\n    X_train_scaled, y_train,\n    validation_data=(X_val_scaled, y_val),\n    epochs=20,\n    batch_size=4096,\n    class_weight={0: 1.0, 1: 30.0},\n    callbacks=[early_stop],\n    verbose=1\n)\n\nprint(\"✅ Training done!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:31:13.447027Z","iopub.execute_input":"2026-04-19T21:31:13.447275Z","iopub.status.idle":"2026-04-19T21:33:40.256621Z","shell.execute_reply.started":"2026-04-19T21:31:13.447253Z","shell.execute_reply":"2026-04-19T21:33:40.256100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 15: Plot training results ────────────────────────\n\n# Plot AUC over epochs to see model improvement\nplt.figure(figsize=(10, 4))\n\nplt.plot(history.history['AUC'], label='Train AUC')\nplt.plot(history.history['val_AUC'], label='Val AUC')\nplt.title('Model AUC over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('AUC')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()\n\nprint(f\"✅ Best Val AUC: {max(history.history['val_AUC']):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:34:25.744470Z","iopub.execute_input":"2026-04-19T21:34:25.745076Z","iopub.status.idle":"2026-04-19T21:34:25.947165Z","shell.execute_reply.started":"2026-04-19T21:34:25.745051Z","shell.execute_reply":"2026-04-19T21:34:25.946477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 16: Compare 3 optimizers ─────────────────────────\n# Assignment requires testing: Adam, SGD, RMSprop\n\nresults = {}  # store AUC results for each optimizer\n\nfor opt_name, optimizer in [\n    ('Adam',    keras.optimizers.Adam(learning_rate=0.001)),\n    ('SGD',     keras.optimizers.SGD(learning_rate=0.001)),\n    ('RMSprop', keras.optimizers.RMSprop(learning_rate=0.001))\n]:\n    print(f\"\\n Training with {opt_name}...\")\n    \n    # Build a fresh model for each optimizer\n    m = keras.Sequential([\n        layers.Input(shape=(X_train_scaled.shape[1],)),\n        layers.Dense(256, activation='relu'),\n        layers.Dropout(0.3),\n        layers.Dense(128, activation='relu'),\n        layers.Dropout(0.3),\n        layers.Dense(64, activation='relu'),\n        layers.Dropout(0.2),\n        layers.Dense(1, activation='sigmoid')\n    ])\n    \n    m.compile(optimizer=optimizer,\n              loss='binary_crossentropy',\n              metrics=['AUC'])\n    \n    hist = m.fit(\n        X_train_scaled, y_train,\n        validation_data=(X_val_scaled, y_val),\n        epochs=5,\n        batch_size=4096,\n        class_weight={0: 1.0, 1: 30.0},\n        verbose=0  # silent training\n    )\n    \n    best_auc = max(hist.history['val_AUC'])\n    results[opt_name] = best_auc\n    print(f\"✅ {opt_name} → Val AUC: {best_auc:.4f}\")\n\nprint(\"\\n=== Optimizer Comparison ===\")\nfor name, auc in results.items():\n    print(f\"{name:10s} → AUC: {auc:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:34:44.853113Z","iopub.execute_input":"2026-04-19T21:34:44.853380Z","iopub.status.idle":"2026-04-19T21:36:37.177248Z","shell.execute_reply.started":"2026-04-19T21:34:44.853359Z","shell.execute_reply":"2026-04-19T21:36:37.176556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 17: Final Evaluation on Validation Set ───────────\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_auc_score\nimport numpy as np\n\n# Get predictions from best model (already trained in Cell 14)\ny_pred_prob = model.predict(X_val_scaled, verbose=0)\ny_pred = (y_pred_prob > 0.5).astype(int).flatten()\n\n# AUC Score\nauc = roc_auc_score(y_val, y_pred_prob)\nprint(f\"✅ Final Val AUC: {auc:.4f}\")\n\n# Classification Report\nprint(\"\\n=== Classification Report ===\")\nprint(classification_report(y_val, y_pred, target_names=['No Default', 'Default']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:40:44.028383Z","iopub.execute_input":"2026-04-19T21:40:44.028629Z","iopub.status.idle":"2026-04-19T21:40:53.898175Z","shell.execute_reply.started":"2026-04-19T21:40:44.028608Z","shell.execute_reply":"2026-04-19T21:40:53.897540Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 18: Plot Confusion Matrix ────────────────────────\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix\n\n# Calculate confusion matrix\ncm = confusion_matrix(y_val, y_pred)\n\n# Plot\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=['No Default', 'Default'],\n            yticklabels=['No Default', 'Default'])\nplt.title('Confusion Matrix', fontsize=13)\nplt.ylabel('Actual')\nplt.xlabel('Predicted')\nplt.tight_layout()\nplt.savefig('confusion_matrix.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:41:40.521900Z","iopub.execute_input":"2026-04-19T21:41:40.522181Z","iopub.status.idle":"2026-04-19T21:41:40.754675Z","shell.execute_reply.started":"2026-04-19T21:41:40.522159Z","shell.execute_reply":"2026-04-19T21:41:40.753956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 19: Plot Training History ────────────────────────\n\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# AUC plot\naxes[0].plot(history.history['AUC'], label='Train AUC')\naxes[0].plot(history.history['val_AUC'], label='Val AUC')\naxes[0].set_title('AUC over Epochs')\naxes[0].set_xlabel('Epoch')\naxes[0].set_ylabel('AUC')\naxes[0].legend()\naxes[0].grid(True)\n\n# Loss plot\naxes[1].plot(history.history['loss'], label='Train Loss')\naxes[1].plot(history.history['val_loss'], label='Val Loss')\naxes[1].set_title('Loss over Epochs')\naxes[1].set_xlabel('Epoch')\naxes[1].set_ylabel('Loss')\naxes[1].legend()\naxes[1].grid(True)\n\nplt.tight_layout()\nplt.savefig('training_history.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:41:54.711745Z","iopub.execute_input":"2026-04-19T21:41:54.712056Z","iopub.status.idle":"2026-04-19T21:41:55.153187Z","shell.execute_reply.started":"2026-04-19T21:41:54.712035Z","shell.execute_reply":"2026-04-19T21:41:55.152497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 20: Optimizer Comparison Chart ───────────────────\n\noptimizers = list(results.keys())\naucs = list(results.values())\n\nplt.figure(figsize=(7, 4))\nbars = plt.bar(optimizers, aucs, color=['#3266ad', '#73726c', '#1D9E75'], width=0.4)\n\n# Add value labels on top of each bar\nfor bar, val in zip(bars, aucs):\n    plt.text(bar.get_x() + bar.get_width()/2, \n             bar.get_height() + 0.001,\n             f'{val:.4f}', ha='center', fontsize=11)\n\nplt.title('Optimizer Comparison - Val AUC')\nplt.ylabel('AUC')\nplt.ylim(0.65, 0.80)\nplt.grid(axis='y', alpha=0.4)\nplt.tight_layout()\nplt.savefig('optimizer_comparison.png', dpi=150)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:42:30.322090Z","iopub.execute_input":"2026-04-19T21:42:30.322329Z","iopub.status.idle":"2026-04-19T21:42:30.509186Z","shell.execute_reply.started":"2026-04-19T21:42:30.322309Z","shell.execute_reply":"2026-04-19T21:42:30.508519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 21: Load and Prepare Test Data ───────────────────\n\n# Load test base file (contains case_id but NO target)\ntest_path = '/kaggle/input/competitions/home-credit-credit-risk-model-stability/parquet_files/test/'\n\ndf_test_base = pd.read_parquet(test_path + 'test_base.parquet')\n\n# Load test static features\ndf_test_static = pd.concat([\n    pd.read_parquet(test_path + 'test_static_0_0.parquet'),\n    pd.read_parquet(test_path + 'test_static_0_1.parquet')\n], ignore_index=True)\n\n# Merge base with static on case_id\ndf_test = df_test_base.merge(df_test_static, on='case_id', how='left')\n\nprint(f\"✅ Test data loaded! Shape: {df_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:43:29.919574Z","iopub.execute_input":"2026-04-19T21:43:29.920377Z","iopub.status.idle":"2026-04-19T21:43:29.994624Z","shell.execute_reply.started":"2026-04-19T21:43:29.920353Z","shell.execute_reply":"2026-04-19T21:43:29.993823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 22: Apply Same Preprocessing to Test Data ────────\n\n# Drop same columns we dropped from training data\ndf_test = df_test.drop(columns=[c for c in cols_to_drop if c in df_test.columns])\n\n# Fill missing values\nnum_cols_test = [c for c in num_cols if c in df_test.columns]\ncat_cols_test = [c for c in cat_cols if c in df_test.columns]\n\ndf_test[num_cols_test] = df_test[num_cols_test].fillna(df_test[num_cols_test].median())\nfor col in cat_cols_test:\n    df_test[col] = df_test[col].fillna(df_test[col].mode()[0])\n\n# Encode categorical columns\nfor col in cat_cols_test:\n    df_test[col] = le.fit_transform(df_test[col].astype(str))\n\n# Create same new features\ndf_test['annuity_to_debt']  = df_test['annuity_780A'] / (df_test['totaldebt_9A'] + 1)\ndf_test['settled_to_debt']  = df_test['totalsettled_863A'] / (df_test['totaldebt_9A'] + 1)\ndf_test['credit_to_income'] = df_test['credamount_770A'] / (df_test['maininc_215A'] + 1)\ndf_test['debt_to_credit']   = df_test['currdebt_22A'] / (df_test['credamount_770A'] + 1)\ndf_test['late_pay_ratio']   = df_test['numinstlswithdpd10_728L'] / (df_test['numinstls_657L'] + 1)\n\nprint(f\"✅ Test preprocessing done! Shape: {df_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:43:51.231979Z","iopub.execute_input":"2026-04-19T21:43:51.232245Z","iopub.status.idle":"2026-04-19T21:43:51.283972Z","shell.execute_reply.started":"2026-04-19T21:43:51.232223Z","shell.execute_reply":"2026-04-19T21:43:51.283054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 23: Align Test Columns with Training Columns ─────\n\n# Make sure test has exactly same columns as training (X)\n# Add missing columns as 0, remove extra columns\nX_test = df_test.drop(columns=['case_id', 'date_decision', 'WEEK_NUM'], errors='ignore')\n\n# Keep only columns that exist in training\nX_test = X_test.reindex(columns=X.columns, fill_value=0)\n\nprint(f\"✅ Columns aligned!\")\nprint(f\"X shape: {X.shape[1]} | X_test shape: {X_test.shape[1]}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:44:09.654579Z","iopub.execute_input":"2026-04-19T21:44:09.655149Z","iopub.status.idle":"2026-04-19T21:44:09.663974Z","shell.execute_reply.started":"2026-04-19T21:44:09.655125Z","shell.execute_reply":"2026-04-19T21:44:09.663448Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 24: Scale and Predict ────────────────────────────\n\n# Apply same scaler fitted on training data\nX_test_scaled = scaler.transform(X_test)\n\n# Predict probability of default for each client\ny_test_pred = model.predict(X_test_scaled, verbose=0).flatten()\n\nprint(f\"✅ Predictions done!\")\nprint(f\"Sample predictions: {y_test_pred[:5]}\")\nprint(f\"Mean predicted probability: {y_test_pred.mean():.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:44:22.654565Z","iopub.execute_input":"2026-04-19T21:44:22.655408Z","iopub.status.idle":"2026-04-19T21:44:22.726904Z","shell.execute_reply.started":"2026-04-19T21:44:22.655365Z","shell.execute_reply":"2026-04-19T21:44:22.725945Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 25: Create Submission File ───────────────────────\n\n# Kaggle expects: case_id, score\nsubmission = pd.DataFrame({\n    'case_id': df_test_base['case_id'],\n    'score':   y_test_pred\n})\n\n# Save as CSV\nsubmission.to_csv('submission.csv', index=False)\n\nprint(f\"✅ Submission file created!\")\nprint(f\"Shape: {submission.shape}\")\nprint(f\"\\nSample:\")\nprint(submission.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:45:04.639578Z","iopub.execute_input":"2026-04-19T21:45:04.639854Z","iopub.status.idle":"2026-04-19T21:45:04.647717Z","shell.execute_reply.started":"2026-04-19T21:45:04.639833Z","shell.execute_reply":"2026-04-19T21:45:04.646799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 27: Correlation with Target ──────────────────────\n# Show which features are most related to loan default\n\ncorrelations = df[num_cols].corrwith(df['target']).abs()\ntop10 = correlations.sort_values(ascending=False).head(10)\n\nplt.figure(figsize=(8, 5))\ntop10.plot(kind='bar', color='#3266ad')\nplt.title('Top 10 Features Correlated with Default')\nplt.ylabel('Correlation')\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\nplt.savefig('correlations.png', dpi=150)\nplt.show()\nprint(\"✅ Done!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:53:37.231094Z","iopub.execute_input":"2026-04-19T21:53:37.231327Z","iopub.status.idle":"2026-04-19T21:53:38.808165Z","shell.execute_reply.started":"2026-04-19T21:53:37.231308Z","shell.execute_reply":"2026-04-19T21:53:38.807542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ── Cell 28: Add L2 Regularization to model ───────────────\n# L2 = punishes the model if weights get too large → prevents overfitting\nfrom tensorflow.keras import regularizers\n\nmodel2 = keras.Sequential([\n    layers.Input(shape=(X_train_scaled.shape[1],)),\n    layers.Dense(256, activation='relu',\n                 kernel_regularizer=regularizers.l2(0.001)),\n    layers.Dropout(0.3),\n    layers.Dense(128, activation='relu',\n                 kernel_regularizer=regularizers.l2(0.001)),\n    layers.Dropout(0.3),\n    layers.Dense(64, activation='relu'),\n    layers.Dropout(0.2),\n    layers.Dense(1, activation='sigmoid')\n])\n\n# Learning rate schedule → reduces LR automatically if no improvement\nlr_schedule = keras.callbacks.ReduceLROnPlateau(\n    monitor='val_loss', factor=0.5, patience=2, verbose=1\n)\n\nearly_stop2 = keras.callbacks.EarlyStopping(\n    monitor='val_AUC', patience=3,\n    restore_best_weights=True, mode='max'\n)\n\nmodel2.compile(optimizer='adam',\n               loss='binary_crossentropy',\n               metrics=['AUC'])\n\nhistory2 = model2.fit(\n    X_train_scaled, y_train,\n    validation_data=(X_val_scaled, y_val),\n    epochs=20,\n    batch_size=4096,\n    class_weight={0: 1.0, 1: 30.0},\n    callbacks=[early_stop2, lr_schedule],\n    verbose=1\n)\n\nprint(f\"✅ Best Val AUC: {max(history2.history['val_AUC']):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-19T21:54:03.981470Z","iopub.execute_input":"2026-04-19T21:54:03.981747Z","iopub.status.idle":"2026-04-19T21:56:25.954392Z","shell.execute_reply.started":"2026-04-19T21:54:03.981722Z","shell.execute_reply":"2026-04-19T21:56:25.953783Z"}},"outputs":[],"execution_count":null}]}