{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":50160,"databundleVersionId":7921029}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nBASE_PATH = \"/kaggle/input/competitions/home-credit-credit-risk-model-stability/csv_files/\"\n\nprint(os.listdir(BASE_PATH + \"train\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:53:30.993380Z","iopub.execute_input":"2026-04-18T10:53:30.993726Z","iopub.status.idle":"2026-04-18T10:53:31.002945Z","shell.execute_reply.started":"2026-04-18T10:53:30.993688Z","shell.execute_reply":"2026-04-18T10:53:31.002052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom glob import glob\nimport gc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:53:31.007730Z","iopub.execute_input":"2026-04-18T10:53:31.008034Z","iopub.status.idle":"2026-04-18T10:53:31.340989Z","shell.execute_reply.started":"2026-04-18T10:53:31.008010Z","shell.execute_reply":"2026-04-18T10:53:31.340105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE = \"/kaggle/input/competitions/home-credit-credit-risk-model-stability/csv_files/\"\n\ntrain_base = pd.read_csv(BASE + \"train/train_base.csv\")\ntest_base  = pd.read_csv(BASE + \"test/test_base.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:53:31.342035Z","iopub.execute_input":"2026-04-18T10:53:31.342496Z","iopub.status.idle":"2026-04-18T10:53:31.978420Z","shell.execute_reply.started":"2026-04-18T10:53:31.342457Z","shell.execute_reply":"2026-04-18T10:53:31.977464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_base","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:53:31.979839Z","iopub.execute_input":"2026-04-18T10:53:31.980231Z","iopub.status.idle":"2026-04-18T10:53:31.996244Z","shell.execute_reply.started":"2026-04-18T10:53:31.980170Z","shell.execute_reply":"2026-04-18T10:53:31.995445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_base","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:53:31.997887Z","iopub.execute_input":"2026-04-18T10:53:31.998258Z","iopub.status.idle":"2026-04-18T10:53:32.012162Z","shell.execute_reply.started":"2026-04-18T10:53:31.998233Z","shell.execute_reply":"2026-04-18T10:53:32.011338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport pandas as pd\n\ndef process_files(files, prefix):\n    result = None\n    \n    for i, f in enumerate(files):\n        print(\"Processing:\", f)\n\n        df = pd.read_csv(f, low_memory=False)\n\n        df = df.loc[:, ~df.columns.duplicated()]\n\n        if \"case_id\" not in df.columns:\n            continue\n\n        # اختيار numeric فقط\n        num_cols = [c for c in df.columns if df[c].dtype != \"object\" and c != \"case_id\"]\n        cols = [\"case_id\"] + num_cols[:10]\n\n        df = df[cols]\n\n        # aggregation\n        agg = df.groupby(\"case_id\").agg([\"mean\", \"max\"])\n\n        # flatten columns\n        agg.columns = [f\"{prefix}_{i}_{c[0]}_{c[1]}\" for c in agg.columns]\n        agg = agg.reset_index()\n\n        if result is None:\n            result = agg\n        else:\n            result = result.merge(agg, on=\"case_id\", how=\"outer\")\n\n        del df, agg\n        gc.collect()\n\n    return result","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:53:32.013290Z","iopub.execute_input":"2026-04-18T10:53:32.013605Z","iopub.status.idle":"2026-04-18T10:53:32.025521Z","shell.execute_reply.started":"2026-04-18T10:53:32.013572Z","shell.execute_reply":"2026-04-18T10:53:32.024764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from glob import glob\n\n# static\ntrain_static = process_files(\n    glob(BASE + \"train/train_static_0_*.csv\"),\n    prefix=\"static\"\n)\n\ntest_static = process_files(\n    glob(BASE + \"test/test_static_0_*.csv\"),\n    prefix=\"static\"\n)\n\n# credit bureau\ntrain_cb = process_files(\n    glob(BASE + \"train/train_credit_bureau_a_1_*.csv\"),\n    prefix=\"cb\"\n)\n\ntest_cb = process_files(\n    glob(BASE + \"test/test_credit_bureau_a_1_*.csv\"),\n    prefix=\"cb\"\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:53:32.026603Z","iopub.execute_input":"2026-04-18T10:53:32.026931Z","iopub.status.idle":"2026-04-18T10:57:06.531021Z","shell.execute_reply.started":"2026-04-18T10:53:32.026896Z","shell.execute_reply":"2026-04-18T10:57:06.530166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train_base.copy()\ntest  = test_base.copy()\n\nfor df in [train_static, train_cb]:\n    train = train.merge(df, on=\"case_id\", how=\"left\")\n    test  = test.merge(df, on=\"case_id\", how=\"left\")\n\ndf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:06.532799Z","iopub.execute_input":"2026-04-18T10:57:06.533052Z","iopub.status.idle":"2026-04-18T10:57:10.036217Z","shell.execute_reply.started":"2026-04-18T10:57:06.533029Z","shell.execute_reply":"2026-04-18T10:57:10.035479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fix_missing(df):\n    missing_cols = {\n        col + \"_missing\": df[col].isna().astype(\"int8\")\n        for col in df.columns\n        if col != \"case_id\"\n    }\n    \n    df_missing = pd.DataFrame(missing_cols)\n    \n    df = pd.concat([df, df_missing], axis=1)\n    \n    df = df.fillna(-1)\n    \n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:10.037265Z","iopub.execute_input":"2026-04-18T10:57:10.037603Z","iopub.status.idle":"2026-04-18T10:57:10.042431Z","shell.execute_reply.started":"2026-04-18T10:57:10.037569Z","shell.execute_reply":"2026-04-18T10:57:10.041783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.copy()\ntest = test.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:10.043271Z","iopub.execute_input":"2026-04-18T10:57:10.043557Z","iopub.status.idle":"2026-04-18T10:57:11.614704Z","shell.execute_reply.started":"2026-04-18T10:57:10.043523Z","shell.execute_reply":"2026-04-18T10:57:11.613907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:11.618104Z","iopub.execute_input":"2026-04-18T10:57:11.618402Z","iopub.status.idle":"2026-04-18T10:57:11.624094Z","shell.execute_reply.started":"2026-04-18T10:57:11.618377Z","shell.execute_reply":"2026-04-18T10:57:11.623513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:11.624925Z","iopub.execute_input":"2026-04-18T10:57:11.625274Z","iopub.status.idle":"2026-04-18T10:57:11.846816Z","shell.execute_reply.started":"2026-04-18T10:57:11.625248Z","shell.execute_reply":"2026-04-18T10:57:11.845863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:11.848098Z","iopub.execute_input":"2026-04-18T10:57:11.848466Z","iopub.status.idle":"2026-04-18T10:57:12.017584Z","shell.execute_reply.started":"2026-04-18T10:57:11.848415Z","shell.execute_reply":"2026-04-18T10:57:12.016700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_df = df.drop(columns=[\"case_id\"]).isna().astype(\"int8\")\nmissing_df.columns = [c + \"_missing\" for c in missing_df.columns]\n\ndf = pd.concat([df, missing_df], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:12.018616Z","iopub.execute_input":"2026-04-18T10:57:12.018940Z","iopub.status.idle":"2026-04-18T10:57:13.520605Z","shell.execute_reply.started":"2026-04-18T10:57:12.018915Z","shell.execute_reply":"2026-04-18T10:57:13.519592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.fillna(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:13.521826Z","iopub.execute_input":"2026-04-18T10:57:13.522315Z","iopub.status.idle":"2026-04-18T10:57:13.984200Z","shell.execute_reply.started":"2026-04-18T10:57:13.522273Z","shell.execute_reply":"2026-04-18T10:57:13.983195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:13.985425Z","iopub.execute_input":"2026-04-18T10:57:13.985735Z","iopub.status.idle":"2026-04-18T10:57:17.563164Z","shell.execute_reply.started":"2026-04-18T10:57:13.985705Z","shell.execute_reply":"2026-04-18T10:57:17.562220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:17.564340Z","iopub.execute_input":"2026-04-18T10:57:17.564699Z","iopub.status.idle":"2026-04-18T10:57:17.751186Z","shell.execute_reply.started":"2026-04-18T10:57:17.564661Z","shell.execute_reply":"2026-04-18T10:57:17.750508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\nnumeric_cols = df.select_dtypes(include=[np.number]).columns\n\ntop_cols = df[numeric_cols].var().sort_values(ascending=False).head(20).index\n\nplot_df = df[top_cols].fillna(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:17.752198Z","iopub.execute_input":"2026-04-18T10:57:17.752470Z","iopub.status.idle":"2026-04-18T10:57:19.500487Z","shell.execute_reply.started":"2026-04-18T10:57:17.752447Z","shell.execute_reply":"2026-04-18T10:57:19.499871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(18, 10))\n\nnp.log1p(plot_df).boxplot(rot=90, grid=False)\n\nplt.title(\"Log-Scaled Boxplot (Better Outlier Visibility)\", fontsize=16)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:19.501408Z","iopub.execute_input":"2026-04-18T10:57:19.502055Z","iopub.status.idle":"2026-04-18T10:57:39.425571Z","shell.execute_reply.started":"2026-04-18T10:57:19.502011Z","shell.execute_reply":"2026-04-18T10:57:39.424675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = \"target\"\n\ny = train[target]\ntrain = train.drop(columns=[target])\n\n# حذف missing العالي\nmissing = train.isnull().mean()\ncols = missing[missing < 0.7].index\n\ntrain = train[cols]\ntest  = test[cols.drop(target, errors='ignore')]\n\n# fillna\ntrain = train.fillna(-1)\ntest  = test.fillna(-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:39.426755Z","iopub.execute_input":"2026-04-18T10:57:39.427067Z","iopub.status.idle":"2026-04-18T10:57:40.884052Z","shell.execute_reply.started":"2026-04-18T10:57:39.427032Z","shell.execute_reply":"2026-04-18T10:57:40.883284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in train.columns:\n    if train[col].dtype == \"float64\":\n        train[col] = train[col].astype(\"float32\")\n        test[col]  = test[col].astype(\"float32\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:40.885048Z","iopub.execute_input":"2026-04-18T10:57:40.885483Z","iopub.status.idle":"2026-04-18T10:57:41.056664Z","shell.execute_reply.started":"2026-04-18T10:57:40.885455Z","shell.execute_reply":"2026-04-18T10:57:41.056015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.select_dtypes(include=[np.number])\ntest = test.select_dtypes(include=[np.number])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:41.057534Z","iopub.execute_input":"2026-04-18T10:57:41.057826Z","iopub.status.idle":"2026-04-18T10:57:41.384410Z","shell.execute_reply.started":"2026-04-18T10:57:41.057800Z","shell.execute_reply":"2026-04-18T10:57:41.383754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train.dtypes.value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:41.385400Z","iopub.execute_input":"2026-04-18T10:57:41.385735Z","iopub.status.idle":"2026-04-18T10:57:41.391881Z","shell.execute_reply.started":"2026-04-18T10:57:41.385688Z","shell.execute_reply":"2026-04-18T10:57:41.390851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\n\ntrain = train_base.copy()\ntest = test_base.copy()\n\nfor df in [train_static, train_cb]:\n    train = train.merge(df, on=\"case_id\", how=\"left\")\n    test = test.merge(df, on=\"case_id\", how=\"left\")\n\ny = train[\"target\"]\ntrain = train.drop(columns=[\"target\"])\n\ntrain = train.apply(pd.to_numeric, errors=\"coerce\")\ntest = test.apply(pd.to_numeric, errors=\"coerce\")\n\ntrain = train.fillna(-1).astype(\"float32\")\ntest = test.fillna(-1).astype(\"float32\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:41.392880Z","iopub.execute_input":"2026-04-18T10:57:41.393223Z","iopub.status.idle":"2026-04-18T10:57:58.235057Z","shell.execute_reply.started":"2026-04-18T10:57:41.393188Z","shell.execute_reply":"2026-04-18T10:57:58.234107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(\n    train, y, test_size=0.2, random_state=42, stratify=y\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:58.236295Z","iopub.execute_input":"2026-04-18T10:57:58.236875Z","iopub.status.idle":"2026-04-18T10:57:59.940529Z","shell.execute_reply.started":"2026-04-18T10:57:58.236838Z","shell.execute_reply":"2026-04-18T10:57:59.939732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train_ids = train[\"case_id\"]\n# test_ids = test[\"case_id\"]\n\n# train = train.drop(columns=[\"case_id\"])\n# test = test.drop(columns=[\"case_id\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:59.941737Z","iopub.execute_input":"2026-04-18T10:57:59.942504Z","iopub.status.idle":"2026-04-18T10:57:59.946061Z","shell.execute_reply.started":"2026-04-18T10:57:59.942462Z","shell.execute_reply":"2026-04-18T10:57:59.945123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.preprocessing import RobustScaler\n\n# scaler = RobustScaler()\n\n# train_scaled = scaler.fit_transform(train)\n# test_scaled = scaler.transform(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:59.946998Z","iopub.execute_input":"2026-04-18T10:57:59.947379Z","iopub.status.idle":"2026-04-18T10:57:59.961497Z","shell.execute_reply.started":"2026-04-18T10:57:59.947353Z","shell.execute_reply":"2026-04-18T10:57:59.960602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train = pd.DataFrame(train_scaled, columns=train.columns)\n# test = pd.DataFrame(test_scaled, columns=test.columns)\n# train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:59.962634Z","iopub.execute_input":"2026-04-18T10:57:59.963268Z","iopub.status.idle":"2026-04-18T10:57:59.972661Z","shell.execute_reply.started":"2026-04-18T10:57:59.963243Z","shell.execute_reply":"2026-04-18T10:57:59.971901Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"normalizer = tf.keras.layers.Normalization()\nnormalizer.adapt(X_train.values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:57:59.973485Z","iopub.execute_input":"2026-04-18T10:57:59.973759Z","iopub.status.idle":"2026-04-18T10:58:00.820995Z","shell.execute_reply.started":"2026-04-18T10:57:59.973736Z","shell.execute_reply":"2026-04-18T10:58:00.820123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inputs = tf.keras.Input(shape=(X_train.shape[1],))\nx = normalizer(inputs)\n\nx = tf.keras.layers.Dense(512, activation=\"relu\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Dropout(0.4)(x)\n\nx = tf.keras.layers.Dense(256, activation=\"relu\")(x)\nx = tf.keras.layers.BatchNormalization()(x)\nx = tf.keras.layers.Dropout(0.3)(x)\n\nx = tf.keras.layers.Dense(128, activation=\"relu\")(x)\nx = tf.keras.layers.Dropout(0.2)(x)\n\noutputs = tf.keras.layers.Dense(1, activation=\"sigmoid\")(x)\n\nmodel = tf.keras.Model(inputs, outputs)\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-3),\n    loss=\"binary_crossentropy\",\n    metrics=[tf.keras.metrics.AUC(name=\"auc\")]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:58:00.822609Z","iopub.execute_input":"2026-04-18T10:58:00.822982Z","iopub.status.idle":"2026-04-18T10:58:01.644293Z","shell.execute_reply.started":"2026-04-18T10:58:00.822957Z","shell.execute_reply":"2026-04-18T10:58:01.643608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callbacks = [\n    tf.keras.callbacks.EarlyStopping(\n        monitor=\"val_auc\",\n        patience=5,\n        mode=\"max\",\n        restore_best_weights=True\n    ),\n    tf.keras.callbacks.ReduceLROnPlateau(\n        monitor=\"val_auc\",\n        factor=0.5,\n        patience=2,\n        mode=\"max\",\n        min_lr=1e-6\n    )\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:58:01.645172Z","iopub.execute_input":"2026-04-18T10:58:01.645444Z","iopub.status.idle":"2026-04-18T10:58:01.649946Z","shell.execute_reply.started":"2026-04-18T10:58:01.645421Z","shell.execute_reply":"2026-04-18T10:58:01.649106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    epochs=50,\n    batch_size=1024,\n    callbacks=callbacks,\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:58:01.651022Z","iopub.execute_input":"2026-04-18T10:58:01.651355Z","iopub.status.idle":"2026-04-18T10:59:34.926198Z","shell.execute_reply.started":"2026-04-18T10:58:01.651331Z","shell.execute_reply":"2026-04-18T10:59:34.925425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom lightgbm import LGBMClassifier\nfrom sklearn.metrics import roc_auc_score\n\nmodel = LGBMClassifier(\n    n_estimators=8000,\n    learning_rate=0.01,\n    num_leaves=128,\n    subsample=0.8,\n    colsample_bytree=0.7,\n    random_state=42\n)\n\nmodel.fit(\n    X_train, y_train,\n    eval_set=[(X_val, y_val)],\n    eval_metric=\"auc\",\n    callbacks=[\n        lgb.early_stopping(300),\n        lgb.log_evaluation(100)\n    ]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T10:59:34.927260Z","iopub.execute_input":"2026-04-18T10:59:34.927677Z","iopub.status.idle":"2026-04-18T11:08:31.005582Z","shell.execute_reply.started":"2026-04-18T10:59:34.927650Z","shell.execute_reply":"2026-04-18T11:08:31.004738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = model.predict_proba(X_val)[:, 1]\nprint(\"AUC:\", roc_auc_score(y_val, pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-18T11:08:31.006745Z","iopub.execute_input":"2026-04-18T11:08:31.007770Z","iopub.status.idle":"2026-04-18T11:09:27.906992Z","shell.execute_reply.started":"2026-04-18T11:08:31.007742Z","shell.execute_reply":"2026-04-18T11:09:27.905994Z"}},"outputs":[],"execution_count":null}]}