{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-08T04:46:38.378878Z","iopub.execute_input":"2025-11-08T04:46:38.379211Z","iopub.status.idle":"2025-11-08T04:46:38.724614Z","shell.execute_reply.started":"2025-11-08T04:46:38.379184Z","shell.execute_reply":"2025-11-08T04:46:38.723641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Sample ~1,000,000 rows from training data with complete labels (Balanced dataset)\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport dask.dataframe as dd\nimport numpy as np\n\n\n# PARQUET_DATA_DIR = \"drive/MyDrive/train_data_parquet\"\n# PARQUET_LABEL_DIR = \"drive/MyDrive/train_labels_parquet\"\n\n# PARQUET_DATA_DIR = \"train_data_parquet\"\n# PARQUET_LABEL_DIR = \"train_labels_parquet\"\n\nTRAIN_DATA_PATH = '/kaggle/input/amex-default-prediction/train_data.csv'\nTRAIN_LABELS_PATH = '/kaggle/input/amex-default-prediction/train_labels.csv'\n\nprint(\"Loading labels...\")\n# df_labels = dd.read_parquet(PARQUET_LABEL_DIR, engine=\"pyarrow\").compute()\ndf_labels = pd.read_csv(TRAIN_LABELS_PATH)\n\nprint(\"Label distribution:\")\nprint(df_labels[\"target\"].value_counts())\n\n# separate counts based on target\nn_default = df_labels[\"target\"].sum() # target=1 is default\nn_nondefault = len(df_labels) - n_default\n\nprint(f\"Default customers: {n_default:,}\")\nprint(f\"Non-default customers: {n_nondefault:,}\")\n\n# calculate average rows per customer\nprint(\"Loading small subset to estimate rows per customer...\")\n# df_train_dask = dd.read_parquet(PARQUET_DATA_DIR, engine=\"pyarrow\")\ndf_train_dask = dd.read_csv(TRAIN_DATA_PATH)\nrows_per_customer_est = int(df_train_dask.shape[0].compute() / len(df_labels))\nprint(f\"Estimated rows per customer: {rows_per_customer_est}\")\n\n# 2 million rows max\ntarget_rows = 2_000_000\nmax_customers_total = target_rows // rows_per_customer_est\nmax_customers_each_class = max_customers_total // 2\n\nfrac_default = max_customers_each_class / n_default\nfrac_nondefault = max_customers_each_class / n_nondefault\n\nprint(f\"Sampling {max_customers_each_class:,} customers from each class\")\n\n# randomly sample customers\ndefault_customers = df_labels[df_labels[\"target\"] == 1].sample(\n    frac=frac_default, random_state=42\n)\nnondefault_customers = df_labels[df_labels[\"target\"] == 0].sample(\n    frac=frac_nondefault, random_state=42\n)\n\n# combine sampled customers\nsampled_customers = pd.concat([default_customers, nondefault_customers])\nsampled_customer_list = sampled_customers[\"customer_ID\"].tolist()\n\nprint(\"Filtering training data to sampled customers...\")\ndf_train_sample = df_train_dask[\n    df_train_dask[\"customer_ID\"].isin(sampled_customer_list)\n].compute()\n\nprint(f\"Loaded {len(df_train_sample):,} rows for sampled customers\")\n\n# if larger than 2 million rows, randomly sample down to 2 million\nif len(df_train_sample) > 2_000_000:\n    df_train_sample = df_train_sample.sample(n=2_000_000, random_state=42)\n\n# combine with labels\ndf_sample = pd.merge(df_train_sample, df_labels, on=\"customer_ID\", how=\"left\")\n\n# check results\nprint(\"\\nSampling done!\")\nprint(f\"Shape: {df_sample.shape}\")\nprint(f\"Unique customers: {df_sample['customer_ID'].nunique()}\")\nprint(df_sample[\"target\"].value_counts(normalize=True))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T04:46:38.726258Z","iopub.execute_input":"2025-11-08T04:46:38.726771Z","iopub.status.idle":"2025-11-08T04:52:19.614399Z","shell.execute_reply.started":"2025-11-08T04:46:38.726743Z","shell.execute_reply":"2025-11-08T04:52:19.612350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3 Final check\ntrain_data_sampled = df_sample\n\n# for later use, convert to normal pandas DataFrame (not pyarrow type)\ntrain_data_sampled = train_data_sampled.convert_dtypes(dtype_backend=\"numpy_nullable\").infer_objects()\n\nprint(type( train_data_sampled ))\n\ndisplay(train_data_sampled.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T04:52:19.616660Z","iopub.execute_input":"2025-11-08T04:52:19.616968Z","iopub.status.idle":"2025-11-08T04:53:25.286483Z","shell.execute_reply.started":"2025-11-08T04:52:19.616942Z","shell.execute_reply":"2025-11-08T04:53:25.285212Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4: Data Cleaning\n\nprint(\"\\nShowing missing value percentages for top 20 columns...\")\nmissing_values_perc = (train_data_sampled.isnull().sum() / len(train_data_sampled)) * 100\nmissing_top20 = missing_values_perc.sort_values(ascending=False).head(20)\ndisplay(missing_top20)\n\nprint(\"Starting data cleaning...\")\n\n# ensure date columns are in datetime format\nif \"S_2\" in train_data_sampled.columns:\n    train_data_sampled[\"S_2\"] = pd.to_datetime(train_data_sampled[\"S_2\"], errors=\"coerce\")\n\n# separate numeric and string columns\nnum_cols = train_data_sampled.select_dtypes(include=[\"number\"]).columns\nstr_cols = train_data_sampled.select_dtypes(include=[\"string\", \"object\"]).columns\n\nprint(f\"Original rows: {len(train_data_sampled.columns)}\")\n\n# remove columns with more than 40% missing data\ncols_to_drop = missing_values_perc[missing_values_perc > 40].index.tolist()\ntrain_data_sampled = train_data_sampled.drop(columns=cols_to_drop)\nprint(f\"Dropped {len(cols_to_drop)} columns with more than 40% missing data.\")\n\n# delete any \"unnamed\" columns if exist\ndrop_cols = [c for c in train_data_sampled.columns if \"unnamed\" in c.lower()]\ntrain_data_sampled = train_data_sampled.drop(columns=drop_cols, errors=\"ignore\")\n\nprint(\"Finished data cleaning!\")\nprint(f\"Rows left: {len(train_data_sampled.columns)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T04:53:25.289627Z","iopub.execute_input":"2025-11-08T04:53:25.289930Z","iopub.status.idle":"2025-11-08T04:53:30.120000Z","shell.execute_reply.started":"2025-11-08T04:53:25.289906Z","shell.execute_reply":"2025-11-08T04:53:30.119117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5: Feature Engineering\n\nprint(\"Beginning feature engineering...\")\n\n# sort by customer_ID and date for time series processing\ntrain_data_sampled = train_data_sampled.sort_values(['customer_ID', 'S_2']).reset_index(drop=True)\n\n# data types of numeric columns\nnum_cols = train_data_sampled.select_dtypes(include=['int64', 'float64']).columns.tolist()\nnum_cols = [c for c in num_cols if c not in ['target']]  # exclude target column\n\n# calculate customer-level aggregate features for numeric columns\nagg_funcs = ['mean', 'max', 'min', 'last']\ncustomer_features = (\n    train_data_sampled\n    .groupby('customer_ID')[num_cols]\n    .agg(agg_funcs)\n)\n\n# flatten multi-level columns\ncustomer_features.columns = ['_'.join(col).strip() for col in customer_features.columns.values]\n\n# combine target\ncustomer_features = customer_features.merge(\n    train_data_sampled[['customer_ID', 'target']].drop_duplicates(),\n    on='customer_ID', how='left'\n)\n\nprint(\"Finished feature engineering!\")\ndisplay(customer_features.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T04:53:30.120822Z","iopub.execute_input":"2025-11-08T04:53:30.121051Z","iopub.status.idle":"2025-11-08T04:53:51.750732Z","shell.execute_reply.started":"2025-11-08T04:53:30.121033Z","shell.execute_reply":"2025-11-08T04:53:51.749629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6: Exploratory Data Analysis (EDA)\n\n# use a cleaner visual theme\nsns.set_style('whitegrid')\n\n# current working DataFrame\ndf = customer_features.dropna().reset_index(drop=True)\n\nprint(f\"After dropping missing values: {len(df)} rows left, {len(df.columns)} columns.\")\n\n# change to parquet\ndf.to_parquet(\"cleaned_data.parquet\", index=False)\n\nprint(\"Summary statistics for numeric columns\")\ndisplay(df.describe().T)\n\nprint(\"\\nListing missing value percentages for top 20 columns...\")\nmissing_values_perc = (df.isnull().sum() / len(df)) * 100\nmissing_top20 = missing_values_perc.sort_values(ascending=False).head(20)\ndisplay(missing_top20)\n\nplt.figure(figsize=(10,6))\nsns.barplot(x=missing_top20.values, y=missing_top20.index, palette='viridis')\nplt.title(\"Top 20 Columns with Most Missing Values\")\nplt.xlabel(\"Missing Value Percentage (%)\")\nplt.ylabel(\"Feature\")\nplt.show()\n\nprint(\"\\nTarget variable distribution (per unique customer)\")\n\nunique_customers = df.drop_duplicates(subset=['customer_ID'])\nplt.figure(figsize=(8,5))\nsns.countplot(x='target', data=unique_customers, palette='pastel')\nplt.title('Distribution of Target Variable (Per Unique Customer)')\nplt.xlabel('Default (1) vs. No Default (0)')\nplt.ylabel('Number of Unique Customers')\n\n# show percentage on top of bars\ntotal = len(unique_customers)\nfor p in plt.gca().patches:\n    height = p.get_height()\n    plt.gca().text(\n        p.get_x() + p.get_width()/2., height + 50,\n        f'{100*height/total:.2f}%', ha='center', fontsize=10\n    )\nplt.show()\n\nprint(\"\\nKey categorical feature count plots\")\nkey_cat_features = ['B_30', 'B_38', 'D_63', 'D_64', 'D_68']\n\nfor col in key_cat_features:\n    if col in df.columns:\n        plt.figure(figsize=(10,5))\n        sns.countplot(y=col, data=df, order=df[col].value_counts().index, palette='coolwarm')\n        plt.title(f'Count Plot for {col}')\n        plt.xscale('log')  # use log scale for better visibility\n        plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T04:53:51.751973Z","iopub.execute_input":"2025-11-08T04:53:51.752472Z","iopub.status.idle":"2025-11-08T04:54:05.887286Z","shell.execute_reply.started":"2025-11-08T04:53:51.752429Z","shell.execute_reply":"2025-11-08T04:54:05.885993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 7: Test data preparation\nimport pandas as pd\nimport os\nimport pyarrow.parquet as pq\nimport pyarrow as pa\n\n# Clean up old parquet dirs if exist\nos.system(\"rm -rf test_data_parquet1\")\nos.makedirs(\"test_data_parquet1\", exist_ok=True)\n\nTEST_DATA_DIR = '/kaggle/input/amex-default-prediction/test_data.csv'\n\n# separatedly load csv and transfer to parquet\nchunksize = 500_000  # 500k rows per chunk\ni = 0\nfor chunk in pd.read_csv(TEST_DATA_DIR, chunksize=chunksize):\n    table = pa.Table.from_pandas(chunk)\n    pq.write_table(table, f\"test_data_parquet1/part_{i}.parquet\", compression=\"snappy\")\n    i += 1\n    print(f\"✅ Wrote data chunk {i}\")\n\n\nprint(\"💾 All train set CSV chunks saved as parquet successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T04:54:05.888420Z","iopub.execute_input":"2025-11-08T04:54:05.889237Z","iopub.status.idle":"2025-11-08T05:10:20.211133Z","shell.execute_reply.started":"2025-11-08T04:54:05.889204Z","shell.execute_reply":"2025-11-08T05:10:20.207839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import dask.dataframe as dd\nimport pandas as pd\nimport numpy as np\nimport gc\nimport os\nimport shutil\n\nPARQUET_TEST = \"/kaggle/working/test_data_parquet1\"\nOUTPUT_FILE = \"/kaggle/working/cleaned_test_data.parquet\"\n\nprint(\"📥 Loading metadata...\")\ndf_meta = dd.read_parquet(PARQUET_TEST, engine=\"pyarrow\")\nnparts = df_meta.npartitions\nprint(f\"Found {nparts} partitions\")\n\nagg_funcs = [\"mean\", \"max\", \"min\", \"last\"]\n\n# Detect columns to drop\nsample = df_meta.get_partition(0).head(10_000, compute=True)\ncols_to_drop = sample.columns[sample.isnull().mean() > 0.4].tolist()\nprint(f\"Dropping {len(cols_to_drop)} high-missing columns\")\n\ndf_final = []\n\nfor i in range(nparts):\n    print(f\"\\n🧩 Processing partition {i+1}/{nparts}\")\n    df = df_meta.get_partition(i).compute()\n    df = df.drop(columns=cols_to_drop, errors=\"ignore\")\n\n    if \"S_2\" in df.columns:\n        df[\"S_2\"] = pd.to_datetime(df[\"S_2\"], errors=\"coerce\")\n    df = df.sort_values([\"customer_ID\", \"S_2\"])\n\n    # Downcast numeric\n    for col in df.select_dtypes(\"float64\"):\n        df[col] = df[col].astype(\"float32\")\n    for col in df.select_dtypes(\"int64\"):\n        df[col] = df[col].astype(\"int32\")\n\n    num_cols = df.select_dtypes(include=[\"number\"]).columns.tolist()\n    if \"target\" in num_cols:\n        num_cols.remove(\"target\")\n\n    # Aggregate within this partition\n    part_agg = df.groupby(\"customer_ID\")[num_cols].agg(agg_funcs)\n    part_agg.columns = [\"_\".join(c) for c in part_agg.columns.to_flat_index()]\n    df_final.append(part_agg)\n\n    # Clean up memory\n    del df, part_agg\n    gc.collect()\n\nprint(\"\\n🔗 Combining partial results...\")\ndf_all = pd.concat(df_final)\ndel df_final\ngc.collect()\n\ndf_final = df_all.groupby(\"customer_ID\").agg(\"last\")\ndel df_all\ngc.collect()\n\n# Remove any old copy before writing\nif os.path.exists(OUTPUT_FILE):\n    os.remove(OUTPUT_FILE)\n\nOLD_DIR = \"/kaggle/working/agg_partitions\"\n\nif os.path.exists(OLD_DIR):\n    print(f\"🧹 Removing old directory: {OLD_DIR}\")\n    shutil.rmtree(OLD_DIR)\n    print(\"✅ Old directory deleted\")\nelse:\n    print(\"No old agg_partitions directory found\")\n\nprint(f\"💾 Writing final file to {OUTPUT_FILE} ...\")\ndf_final.to_parquet(OUTPUT_FILE, index=True)\n\nprint(\"\\n🎉 Final features saved successfully!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T05:10:20.215471Z","iopub.execute_input":"2025-11-08T05:10:20.216482Z","iopub.status.idle":"2025-11-08T05:15:11.260682Z","shell.execute_reply.started":"2025-11-08T05:10:20.216416Z","shell.execute_reply":"2025-11-08T05:15:11.259026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import KFold\nfrom sklearn.linear_model import LogisticRegression\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier\nimport optuna\nfrom sklearn.metrics import accuracy_score, f1_score\nimport dask.dataframe as dd\n\ndef amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)\n\n# Load dataset lazily with Dask\nddf = dd.read_parquet('/kaggle/working/cleaned_data.parquet')\n\n# Define feature columns (exclude unnecessary columns)\nfeature_cols = [c for c in ddf.columns if c not in ['customer_ID', 'target']]\n\n# Sample fraction for model selection (including target)\nsample_frac = 1\nddf_sample = ddf[feature_cols + ['target']].sample(frac=sample_frac, random_state=42).compute()\n\n# Split features and target\nX_sample = ddf_sample[feature_cols]\ny_sample = ddf_sample['target']\n\n# 5-fold cross-validation\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n#study = optuna.create_study(direction=\"maximize\")\n\ndef objective(trial):\n  # Determine hyperparameter values\n  learning_rate = trial.suggest_float(\"learning_rate\", 0.01, 0.1)\n  num_leaves = trial.suggest_int(\"num_leaves\", 2, 256)\n  max_depth = trial.suggest_int(\"max_depth\", 5, 30)\n  min_child_samples = trial.suggest_int(\"min_child_samples\", 5, 100)\n  subsample = trial.suggest_float(\"subsample\", 0.5, 1.0)\n  colsample_bytree = trial.suggest_float(\"colsample_bytree\", 0.5, 1.0)\n  n_estimators = trial.suggest_int(\"n_estimators\", 100, 1000)\n    \n  model = LGBMClassifier(\n    learning_rate=learning_rate,\n    num_leaves=num_leaves,\n    max_depth=max_depth,\n    min_child_samples=min_child_samples,\n    subsample=subsample,\n    colsample_bytree=colsample_bytree,\n    n_estimators=n_estimators,\n    random_state=42\n  )\n  amex_scores = []\n  for train_idx, test_idx in kf.split(X_sample):\n        X_train, X_test = X_sample.iloc[train_idx], X_sample.iloc[test_idx]\n        y_train, y_test = y_sample.iloc[train_idx], y_sample.iloc[test_idx]\n\n        # Fit model\n        model.fit(X_train, y_train)\n\n        # Predict\n        y_pred_proba = model.predict_proba(X_test)[:, 1]\n        y_test_df = pd.DataFrame({'target': y_test.values})\n        y_pred_df = pd.DataFrame({'prediction': y_pred_proba})\n      \n        amex_scores.append(amex_metric(y_test_df, y_pred_df))\n\n  return np.mean(amex_scores)\n\n# Run the study and review the results\n#study.optimize(objective, n_trials=20)\n#print(\"Best trial:\")\n#print(\" Value: {}\".format(study.best_trial.value))\n#print(\" Params: {}\".format(study.best_trial.params))\nparams = {\n    \"learning_rate\": 0.011326681203182443,\n    \"num_leaves\": 76,\n    \"max_depth\": -1, \n    \"min_child_samples\": 70,\n    \"subsample\": 0.638068300141083,\n    \"colsample_bytree\": 0.639335047549834,\n    \"n_estimators\": 915\n}\n\n# Define models\nmodels = [\n    ('LogisticRegression', LogisticRegression(max_iter=1000)),\n    ('LightGBM', LGBMClassifier()),\n    ('LightGBM-tuned', LGBMClassifier(**params)),\n    ('XGBoost', XGBClassifier(use_label_encoder=False, eval_metric='logloss'))\n]\n\n# 5-fold cross-validation\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\n\n# Function to evaluate a model\ndef evaluate_model(model, X, y, kf):\n    acc_scores = []\n    f1_scores = []\n    amex_scores = []\n\n    for train_idx, test_idx in kf.split(X):\n        X_train, X_test = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_test = y.iloc[train_idx], y.iloc[test_idx]\n\n        # Fit model\n        model.fit(X_train, y_train)\n\n        # Predict\n        y_pred = model.predict(X_test)\n        y_pred_proba = model.predict_proba(X_test)[:, 1]\n\n        # Compute metrics\n        acc_scores.append(accuracy_score(y_test, y_pred))\n        f1_scores.append(f1_score(y_test, y_pred))\n\n        # amex score\n        y_test_df = pd.DataFrame({'target': y_test.values})\n        y_pred_df = pd.DataFrame({'prediction': y_pred_proba})\n        amex_scores.append(amex_metric(y_test_df, y_pred_df))\n        \n    return np.mean(acc_scores), np.mean(f1_scores), np.mean(amex_scores)\n\n# Evaluate all models\nbest_model_name = None\nbest_amex = 0\nresults = []\n\nfor name, model in models:\n    print(f\"Evaluating {name} ...\")\n    acc, f1, amex = evaluate_model(model, X_sample, y_sample, kf)\n    results.append((name, acc, f1, amex))\n    print(f\"{name}: avg accuracy={acc:.4f}, avg f1={f1:.4f}, avg amex={amex:.4f}\")\n\n    if amex > best_amex:\n        best_amex = amex\n        best_model_name = name\n\n# Summary\nprint(\"\\nSummary of all models:\")\nfor name, acc, f1, amex in results:\n    print(f\"{name}: accuracy={acc:.4f}, f1={f1:.4f}, amex={amex:.4f}\")\n\nprint(f\"\\nBest model based on AMEX score: {best_model_name} with F1={best_amex:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T05:53:02.666591Z","iopub.execute_input":"2025-11-08T05:53:02.666971Z","iopub.status.idle":"2025-11-08T06:23:53.028916Z","shell.execute_reply.started":"2025-11-08T05:53:02.666942Z","shell.execute_reply":"2025-11-08T06:23:53.027591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n# Summary\nprint(\"\\nSummary of all models:\")\nfor name, acc, f1, amex in results:\n    print(f\"{name}: accuracy={acc:.4f}, f1={f1:.4f}, amex={amex:.4f}\")\n\nprint(f\"\\nBest model based on AMEX score: {best_model_name} with F1={best_amex:.4f}\")\n\ntest_data = pd.read_parquet('/kaggle/working/cleaned_test_data.parquet')\ntest_data = test_data.reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:23:53.030775Z","iopub.execute_input":"2025-11-08T06:23:53.031454Z","iopub.status.idle":"2025-11-08T06:24:40.950755Z","shell.execute_reply.started":"2025-11-08T06:23:53.031418Z","shell.execute_reply":"2025-11-08T06:24:40.948613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(test_data.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:24:40.953112Z","iopub.execute_input":"2025-11-08T06:24:40.953576Z","iopub.status.idle":"2025-11-08T06:24:40.982959Z","shell.execute_reply.started":"2025-11-08T06:24:40.953526Z","shell.execute_reply":"2025-11-08T06:24:40.981924Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(ddf.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:24:40.986216Z","iopub.execute_input":"2025-11-08T06:24:40.986590Z","iopub.status.idle":"2025-11-08T06:24:48.158568Z","shell.execute_reply.started":"2025-11-08T06:24:40.986564Z","shell.execute_reply":"2025-11-08T06:24:48.157264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"diff = list(set(test_data.columns)-set(ddf.columns))\nprint(diff)\ntest_data = test_data.drop(columns=diff, axis=1)\ntest_feature_cols = test_data.drop('customer_ID', axis=1).columns\nprint(test_feature_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:24:48.159469Z","iopub.execute_input":"2025-11-08T06:24:48.159838Z","iopub.status.idle":"2025-11-08T06:24:51.746766Z","shell.execute_reply.started":"2025-11-08T06:24:48.159807Z","shell.execute_reply":"2025-11-08T06:24:51.745574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_dict = dict(models)\nfinal_model = model_dict[best_model_name]\nX, y = ddf[feature_cols], ddf['target']\nfinal_model.fit(X.compute(), y.compute())\nX_test = test_data[test_feature_cols]\n# Feed transformed test data into the final_model prediction to get our output values\nprobs = final_model.predict_proba(X_test)[:, 1]\n\n# Create the final submission csv using the model predictions\nsubmission = pd.DataFrame({'customer_ID': test_data['customer_ID'], 'prediction': probs})\nsubmission = submission.set_index('customer_ID')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:24:51.747999Z","iopub.execute_input":"2025-11-08T06:24:51.748677Z","iopub.status.idle":"2025-11-08T06:30:25.689937Z","shell.execute_reply.started":"2025-11-08T06:24:51.748631Z","shell.execute_reply":"2025-11-08T06:30:25.688418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T06:30:25.691181Z","iopub.execute_input":"2025-11-08T06:30:25.691495Z","iopub.status.idle":"2025-11-08T06:30:29.073813Z","shell.execute_reply.started":"2025-11-08T06:30:25.691470Z","shell.execute_reply":"2025-11-08T06:30:29.072401Z"}},"outputs":[],"execution_count":null}]}