{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"},{"sourceId":3727003,"sourceType":"datasetVersion","datasetId":2213609}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T01:02:51.065054Z","iopub.execute_input":"2025-04-16T01:02:51.065452Z","iopub.status.idle":"2025-04-16T01:02:51.084913Z","shell.execute_reply.started":"2025-04-16T01:02:51.065417Z","shell.execute_reply":"2025-04-16T01:02:51.083846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load feather files\ntrain_data = pd.read_feather('/kaggle/input/amexfeather/train_data.ftr')\ntest_data = pd.read_feather('/kaggle/input/amexfeather/test_data.ftr')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T01:02:51.086232Z","iopub.execute_input":"2025-04-16T01:02:51.086589Z","iopub.status.idle":"2025-04-16T01:03:08.457728Z","shell.execute_reply.started":"2025-04-16T01:02:51.086556Z","shell.execute_reply":"2025-04-16T01:03:08.456643Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data_cols = train_data.columns.tolist()\n#print(\"train data columns -- \")\n#print(train_data_cols)\ntest_data_cols = train_data.columns.tolist()\n#print(\"test data columns -- \")\n#print(test_data_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T01:03:08.459015Z","iopub.execute_input":"2025-04-16T01:03:08.459326Z","iopub.status.idle":"2025-04-16T01:03:08.464283Z","shell.execute_reply.started":"2025-04-16T01:03:08.459294Z","shell.execute_reply":"2025-04-16T01:03:08.463402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======================\n# 1. Feature Aggregation\n# ======================\n\n\n# Define aggregations for key features\naggs = {\n    'P_2': ['mean', 'last', 'max', 'min', 'std'],\n    'D_39': ['sum', 'mean'],\n    'B_1': ['mean', 'max', 'min'],\n    'S_2': [lambda x: (x.max() - x.min()).days],\n    'D_41': ['sum', 'mean'],\n    'R_1': ['mean', 'max']\n}\n\n# Aggregate time-series data per customer\ntrain_agg = train_data.groupby('customer_ID').agg(aggs)\ntest_agg = test_data.groupby('customer_ID').agg(aggs)\n\n# Flatten multi-index columns\ntrain_agg.columns = ['_'.join(col) for col in train_agg.columns]\ntest_agg.columns = ['_'.join(col) for col in test_agg.columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T01:03:08.465415Z","iopub.execute_input":"2025-04-16T01:03:08.465764Z","iopub.status.idle":"2025-04-16T01:06:09.983584Z","shell.execute_reply.started":"2025-04-16T01:03:08.465732Z","shell.execute_reply":"2025-04-16T01:06:09.982168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======================\n# 2. Prepare Targets\n# ======================\n# Extract target from train data\ny_train = train_data.groupby('customer_ID')['target'].first()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T01:06:09.985737Z","iopub.execute_input":"2025-04-16T01:06:09.986159Z","iopub.status.idle":"2025-04-16T01:06:11.326977Z","shell.execute_reply.started":"2025-04-16T01:06:09.986126Z","shell.execute_reply":"2025-04-16T01:06:11.325737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======================\n# 3. Train/Validation Split\n# ======================\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train_split, y_val = train_test_split(\n    train_agg, \n    y_train,\n    test_size=0.2,\n    stratify=y_train,\n    random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T01:06:11.329650Z","iopub.execute_input":"2025-04-16T01:06:11.329968Z","iopub.status.idle":"2025-04-16T01:06:11.788351Z","shell.execute_reply.started":"2025-04-16T01:06:11.329944Z","shell.execute_reply":"2025-04-16T01:06:11.787304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======================\n# 4. LightGBM Setup\n# ======================\nimport lightgbm as lgb\n\n# Sample weights for class imbalance\nsample_weights = np.where(y_train_split == 0, 20, 1)\n\n# Define custom metric (must match competition metric)\ndef amex_metric_lgb(y_true, y_pred):\n    import numpy as np\n    import pandas as pd\n\n    df = pd.DataFrame({'target': y_true, 'prediction': y_pred})\n\n    # 1. Calculate Default Rate at 4% (D)\n    df = df.sort_values('prediction', ascending=False).reset_index(drop=True)\n    df['weight'] = np.where(df['target'] == 0, 20, 1)\n    four_pct_cutoff = int(0.04 * df['weight'].sum())\n    df['weight_cumsum'] = df['weight'].cumsum()\n    df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n    d = df_cutoff['target'].sum() / df['target'].sum()\n\n    # 2. Calculate Normalized Gini Coefficient (G)\n    df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n    total_pos = (df['target'] * df['weight']).sum()\n    df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n    df['lorentz'] = df['cum_pos_found'] / total_pos\n    df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n    g = df['gini'].sum()\n\n    # 3. Normalize against perfect model\n    g_max = weighted_gini(df['target'], df['target'])\n    g_normalized = g / g_max if g_max != 0 else 0\n\n    # 4. Combine metrics\n    m = 0.5 * (g_normalized + d)\n\n    return 'amex_metric', m, True\n\ndef weighted_gini(y_true, y_pred):\n    import numpy as np\n    import pandas as pd\n    df = pd.DataFrame({'target': y_true, 'prediction': y_pred})\n    df = df.sort_values('prediction', ascending=False)\n    df['weight'] = np.where(df['target'] == 0, 20, 1)\n    df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n    total_pos = (df['target'] * df['weight']).sum()\n    df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n    df['lorentz'] = df['cum_pos_found'] / total_pos\n    df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n    return df['gini'].sum()\n\n\n\n# Model parameters\nparams = {\n    'objective': 'binary',\n    'metric': 'custom',\n    'boosting_type': 'gbdt',\n    'learning_rate': 0.05,\n    'num_leaves': 127,\n    'min_child_samples': 2400,\n    'feature_fraction': 0.8,\n    'bagging_freq': 1,\n    'verbosity': -1\n}\n\n# Train model\nmodel = lgb.LGBMClassifier(**params)\nmodel.fit(\n    X_train, y_train_split,\n    sample_weight=sample_weights,\n    eval_set=[(X_val, y_val)],\n    eval_metric=amex_metric_lgb,\n    callbacks=[lgb.early_stopping(100)]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T01:06:11.789238Z","iopub.execute_input":"2025-04-16T01:06:11.789495Z","iopub.status.idle":"2025-04-16T01:06:21.951293Z","shell.execute_reply.started":"2025-04-16T01:06:11.789477Z","shell.execute_reply":"2025-04-16T01:06:21.950321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======================\n# 5. Generate Predictions\n# ======================\n# Test set predictions\ntest_preds = model.predict_proba(test_agg)[:, 1]\n\n# Create submission\nsubmission = pd.DataFrame({\n    'customer_ID': test_agg.index,\n    'prediction': test_preds\n})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T01:06:21.952375Z","iopub.execute_input":"2025-04-16T01:06:21.952710Z","iopub.status.idle":"2025-04-16T01:06:33.946332Z","shell.execute_reply.started":"2025-04-16T01:06:21.952681Z","shell.execute_reply":"2025-04-16T01:06:33.945313Z"}},"outputs":[],"execution_count":null}]}