{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# Use the kagglehub client library to attach Kaggle resources like competitions, datasets, and models to your session\n# Learn more about kagglehub: https://github.com/Kaggle/kagglehub/blob/main/README.md\n\nimport kagglehub\n# kagglehub.dataset_download('<owner>/<dataset-slug>')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:28:19.113468Z","iopub.execute_input":"2026-06-22T11:28:19.113807Z","iopub.status.idle":"2026-06-22T11:28:19.125814Z","shell.execute_reply.started":"2026-06-22T11:28:19.113777Z","shell.execute_reply":"2026-06-22T11:28:19.124788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cài đặt các thư viện cần thiết (chạy 1 lần)\n# Trên Kaggle Notebook / Google Colab, hầu hết đã có sẵn TensorFlow, sklearn, pandas.\n# Cần cài thêm: dask, lightgbm, kaggle (nếu chưa có)\n\n!pip install -q dask[complete] lightgbm kaggle pyarrow","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:28:19.127664Z","iopub.execute_input":"2026-06-22T11:28:19.127960Z","iopub.status.idle":"2026-06-22T11:28:23.787519Z","shell.execute_reply.started":"2026-06-22T11:28:19.127934Z","shell.execute_reply":"2026-06-22T11:28:23.786242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport json\nimport time\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport numpy as np\nimport pandas as pd\nimport dask.dataframe as dd\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.decomposition import PCA, TruncatedSVD\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (\n    roc_auc_score, classification_report, confusion_matrix,\n    precision_recall_curve, average_precision_score\n)\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, callbacks, regularizers\n\nSEED = 42\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\nprint(\"TensorFlow version:\", tf.__version__)\nprint(\"GPU available:\", tf.config.list_physical_devices('GPU'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:28:23.789220Z","iopub.execute_input":"2026-06-22T11:28:23.789577Z","iopub.status.idle":"2026-06-22T11:28:33.960102Z","shell.execute_reply.started":"2026-06-22T11:28:23.789540Z","shell.execute_reply":"2026-06-22T11:28:33.959216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport dask.dataframe as dd\n\nDATA_DIR = \"/kaggle/input/competitions/talkingdata-adtracking-fraud-detection\"\n\nprint(\"Các file trong thư mục dữ liệu:\")\nprint(os.listdir(DATA_DIR))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:28:33.967802Z","iopub.execute_input":"2026-06-22T11:28:33.968323Z","iopub.status.idle":"2026-06-22T11:28:34.081132Z","shell.execute_reply.started":"2026-06-22T11:28:33.968283Z","shell.execute_reply":"2026-06-22T11:28:34.080183Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gc\n\n# 1. Định nghĩa kiểu dữ liệu để tiết kiệm tối đa bộ nhớ\ndtypes = {\n    'ip': 'uint32', 'app': 'uint16', 'device': 'uint16', \n    'os': 'uint16', 'channel': 'uint16', 'is_attributed': 'uint8'\n}\n\nfile_path = '/kaggle/input/competitions/talkingdata-adtracking-fraud-detection/train.csv'\n\n# 2. Đọc dữ liệu (nrows=10000000 để giới hạn 10 triệu dòng, tránh tràn RAM)\n# Nếu vẫn tràn, hãy giảm xuống còn 5000000\ndf = pd.read_csv(file_path, dtype=dtypes, usecols=['ip', 'app', 'channel', 'click_time', 'is_attributed'], nrows=10000000)\n\n# 3. Xử lý thời gian\ndf['click_time'] = pd.to_datetime(df['click_time'])\ndf['hour'] = df['click_time'].dt.hour.astype('uint8')\ndf['day'] = df['click_time'].dt.day.astype('uint8')\n\n# 4. Tạo biến đếm (Sử dụng del để giải phóng bộ nhớ ngay lập tức)\nip_count = df.groupby('ip')['app'].count().astype('uint32').rename('ip_count')\ndf = df.merge(ip_count, on='ip', how='left')\ndel ip_count\ngc.collect() # <--- Lệnh dọn dẹp RAM \"thần thánh\"\n\nip_app_count = df.groupby(['ip', 'app'])['channel'].count().astype('uint32').rename('ip_app_count')\ndf = df.merge(ip_app_count, on=['ip', 'app'], how='left')\ndel ip_app_count\ngc.collect()\n\n# 5. Thời gian giữa các lần click\ndf = df.sort_values(['ip', 'click_time'])\ndf['time_since_prev_click'] = df.groupby('ip')['click_time'].diff().dt.total_seconds().fillna(0).astype('float32')\n\n# Dọn dẹp cuối\ndf = df.drop(columns=['click_time'])\ngc.collect()\n\nprint(\"✅ Đã xử lý xong dữ liệu mà không bị tràn RAM!\")\ndisplay(df.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:28:34.083917Z","iopub.execute_input":"2026-06-22T11:28:34.084297Z","iopub.status.idle":"2026-06-22T11:28:54.067841Z","shell.execute_reply.started":"2026-06-22T11:28:34.084261Z","shell.execute_reply":"2026-06-22T11:28:54.067027Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 6))\n\n# Vẽ biểu đồ so sánh giữa người dùng thật và bot\n# hue='is_attributed' giúp Bebbi thấy rõ sự khác biệt hành vi giữa 2 nhóm này\nsns.countplot(x='hour', hue='is_attributed', data=df, palette='viridis')\n\nplt.title('Phân bổ số lượng click theo giờ (So sánh giữa Click thật và Click ảo)', fontsize=15)\nplt.xlabel('Giờ trong ngày', fontsize=12)\nplt.ylabel('Số lượng click', fontsize=12)\nplt.legend(title='Kết quả (0=Ảo, 1=Thật)')\nplt.grid(axis='y', linestyle='--', alpha=0.6)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:28:54.069142Z","iopub.execute_input":"2026-06-22T11:28:54.069406Z","iopub.status.idle":"2026-06-22T11:29:12.022393Z","shell.execute_reply.started":"2026-06-22T11:28:54.069380Z","shell.execute_reply":"2026-06-22T11:29:12.021454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#LÀM SẠCH DỮ LIỆU (DATA CLEANING)\n# ============================================\n\n# 1. Liệt kê các cột gốc không còn cần thiết sau khi đã trích xuất đặc trưng\n# 'click_time' và 'attributed_time' là dạng thời gian thô, không dùng trực tiếp cho mô hình\ncols_to_drop = ['click_time', 'attributed_time']\n\n# 2. Loại bỏ các cột không dùng đến\n# Kiểm tra xem cột có tồn tại trong df không rồi mới drop để tránh lỗi\ndf_clean = df.drop(columns=[c for c in cols_to_drop if c in df.columns])\n\n# 3. Loại bỏ các dòng có giá trị NaN (nếu có)\n# Các bước tạo feature như diff() thường tạo ra NaN ở dòng đầu tiên\ndf_clean = df_clean.dropna()\n\n# 4. Kiểm tra lại thông tin sau khi làm sạch\nprint(\"--- Kết quả sau khi làm sạch ---\")\nprint(f\"Số cột còn lại: {df_clean.shape[1]}\")\nprint(f\"Danh sách cột: {df_clean.columns.tolist()}\")\n\n# 5. Lưu thành file 'talkingdata_clean.csv'\ndf_clean.to_csv('talkingdata_clean.csv', index=False)\nprint(\"\\n✅ File 'talkingdata_clean.csv' đã được lưu thành công!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:29:12.023647Z","iopub.execute_input":"2026-06-22T11:29:12.023999Z","iopub.status.idle":"2026-06-22T11:29:47.761656Z","shell.execute_reply.started":"2026-06-22T11:29:12.023960Z","shell.execute_reply":"2026-06-22T11:29:47.760585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load dữ liệu sạch\ndf = pd.read_csv('talkingdata_clean.csv')\n\n# Tách X và y\nX = df.drop('is_attributed', axis=1)\ny = df['is_attributed']\n\n# Chia train/val (stratify rất quan trọng vì dữ liệu fraud rất ít)\nfrom sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n\n# TẠM DỪNG: Cần thêm bước Scaling (Chuẩn hóa) nếu bạn dùng mô hình Deep Learning (CNN/Autoencoder)\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_val = scaler.transform(X_val)\n\nprint(\"✅ Dữ liệu đã sẵn sàng để train!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:29:47.762806Z","iopub.execute_input":"2026-06-22T11:29:47.763096Z","iopub.status.idle":"2026-06-22T11:30:08.858985Z","shell.execute_reply.started":"2026-06-22T11:29:47.763054Z","shell.execute_reply":"2026-06-22T11:30:08.858087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nprint(\"🌳 Đang huấn luyện Random Forest...\")\nrf_model = RandomForestClassifier(n_estimators=100, max_depth=10, n_jobs=-1, random_state=42)\nrf_model.fit(X_train, y_train)\n\nrf_pred = rf_model.predict_proba(X_val)[:, 1]\nprint(f\"Random Forest AUC: {roc_auc_score(y_val, rf_pred):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:30:08.860591Z","iopub.execute_input":"2026-06-22T11:30:08.860986Z","iopub.status.idle":"2026-06-22T11:43:11.507492Z","shell.execute_reply.started":"2026-06-22T11:30:08.860949Z","shell.execute_reply":"2026-06-22T11:43:11.506514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\nprint(\"🚀 Đang huấn luyện XGBoost...\")\nxgb_model = XGBClassifier(\n    n_estimators=200, \n    learning_rate=0.05, \n    max_depth=6, \n    scale_pos_weight=99, # Quan trọng vì dữ liệu mất cân bằng\n    n_jobs=-1,\n    random_state=42\n)\nxgb_model.fit(X_train, y_train)\n\nxgb_pred = xgb_model.predict_proba(X_val)[:, 1]\nprint(f\"XGBoost AUC: {roc_auc_score(y_val, xgb_pred):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:43:11.509110Z","iopub.execute_input":"2026-06-22T11:43:11.509437Z","iopub.status.idle":"2026-06-22T11:44:30.529534Z","shell.execute_reply.started":"2026-06-22T11:43:11.509409Z","shell.execute_reply":"2026-06-22T11:44:30.528596Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\n\nprint(\"💡 Đang huấn luyện LightGBM...\")\nlgb_model = lgb.LGBMClassifier(\n    n_estimators=500, \n    learning_rate=0.05, \n    is_unbalance=True, # Tự động xử lý dữ liệu mất cân bằng\n    n_jobs=-1,\n    random_state=42\n)\nlgb_model.fit(X_train, y_train)\n\nlgb_pred = lgb_model.predict_proba(X_val)[:, 1]\nprint(f\"LightGBM AUC: {roc_auc_score(y_val, lgb_pred):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:44:30.530911Z","iopub.execute_input":"2026-06-22T11:44:30.531221Z","iopub.status.idle":"2026-06-22T11:46:19.162978Z","shell.execute_reply.started":"2026-06-22T11:44:30.531194Z","shell.execute_reply":"2026-06-22T11:46:19.161939Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Danh sách kết quả (Bạn hãy thay số bằng AUC thực tế bạn nhận được)\nmodel_names = ['Random Forest', 'XGBoost', 'LightGBM']\nauc_scores = [0.9559,  0.9666, 0.9155] # Thay bằng giá trị auc bạn in ra được nhé\n\nplt.figure(figsize=(8, 5))\nplt.bar(model_names, auc_scores, color=['skyblue', 'salmon', 'lightgreen'])\nplt.ylim(0.7, 1.0) # Tập trung vào dải điểm cao\nplt.title('So sánh AUC giữa các mô hình Baseline')\nplt.ylabel('AUC Score')\nplt.grid(axis='y', linestyle='--', alpha=0.7)\n\n# Hiện giá trị lên trên đầu cột\nfor i, v in enumerate(auc_scores):\n    plt.text(i, v + 0.01, f\"{v:.4f}\", ha='center', fontweight='bold')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T16:07:45.507926Z","iopub.execute_input":"2026-06-22T16:07:45.508315Z","iopub.status.idle":"2026-06-22T16:07:45.653359Z","shell.execute_reply.started":"2026-06-22T16:07:45.508285Z","shell.execute_reply":"2026-06-22T16:07:45.652237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\n\n# Giả sử X_train là dữ liệu đã được StandardScaled\ninput_dim = X_train.shape[1] \n\n# Kiến trúc Autoencoder\ninput_layer = layers.Input(shape=(input_dim,))\n# Encoder: Nén dữ liệu\nencoded = layers.Dense(16, activation='relu')(input_layer)\nencoded = layers.Dense(8, activation='relu')(encoded) # Latent space (8 đặc trưng mới)\n\n# Decoder: Tái tạo lại dữ liệu\ndecoded = layers.Dense(16, activation='relu')(encoded)\ndecoded = layers.Dense(input_dim, activation='sigmoid')(decoded)\n\nautoencoder = models.Model(input_layer, decoded)\nencoder = models.Model(input_layer, encoded) # Đây là model chúng ta sẽ dùng để lấy feature\n\nautoencoder.compile(optimizer='adam', loss='mse')\nautoencoder.fit(X_train, X_train, epochs=20, batch_size=256, shuffle=True, validation_split=0.2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T11:46:19.325413Z","iopub.execute_input":"2026-06-22T11:46:19.325720Z","iopub.status.idle":"2026-06-22T12:04:57.122749Z","shell.execute_reply.started":"2026-06-22T11:46:19.325694Z","shell.execute_reply":"2026-06-22T12:04:57.121030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Trích xuất đặc trưng mới\nX_train_encoded = encoder.predict(X_train)\nX_val_encoded = encoder.predict(X_val)\n\n# Vẽ biểu đồ mất mát (Loss) để xem Autoencoder đã học tốt chưa\nplt.figure(figsize=(8, 4))\nplt.plot(autoencoder.history.history['loss'], label='Train Loss')\nplt.plot(autoencoder.history.history['val_loss'], label='Val Loss')\nplt.title('Quá trình học của Autoencoder (Reconstruction Loss)')\nplt.xlabel('Epochs')\nplt.ylabel('Loss (MSE)')\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T12:04:57.129897Z","iopub.execute_input":"2026-06-22T12:04:57.130554Z","iopub.status.idle":"2026-06-22T12:13:42.151300Z","shell.execute_reply.started":"2026-06-22T12:04:57.130495Z","shell.execute_reply":"2026-06-22T12:13:42.150001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Lưu các mảng đã nén thành file .npy\nnp.save('X_train_encoded.npy', X_train_encoded)\nnp.save('X_val_encoded.npy', X_val_encoded)\n\nprint(\"✅ Đã lưu xong dữ liệu nén!\")\nprint(f\"File 'X_train_encoded.npy' có kích thước: {X_train_encoded.shape}\")\nprint(f\"File 'X_val_encoded.npy' có kích thước: {X_val_encoded.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T12:13:42.152985Z","iopub.execute_input":"2026-06-22T12:13:42.153366Z","iopub.status.idle":"2026-06-22T12:13:42.503674Z","shell.execute_reply.started":"2026-06-22T12:13:42.153334Z","shell.execute_reply":"2026-06-22T12:13:42.502622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import roc_auc_score\nfrom joblib import Parallel, delayed\nimport gc\n\n# 1. Load và tối ưu dữ liệu (Đảm bảo dữ liệu là 2D)\nX_train_encoded = np.load('X_train_encoded.npy')\nX_val_encoded = np.load('X_val_encoded.npy')\nif X_train_encoded.ndim == 1: X_train_encoded = X_train_encoded.reshape(-1, 8)\nif X_val_encoded.ndim == 1: X_val_encoded = X_val_encoded.reshape(-1, 8)\n\n# 2. Hàm đánh giá (Fitness) - Tối ưu tốc độ\ndef evaluate_features(mask, X_tr, y_tr, X_v, y_v):\n    if np.sum(mask) == 0: return 0\n    # Dùng subset cố định để mô hình học nhanh hơn\n    model = RandomForestClassifier(n_estimators=10, n_jobs=1, max_depth=5, random_state=42)\n    model.fit(X_tr[:, mask], y_tr)\n    return roc_auc_score(y_v, model.predict_proba(X_v[:, mask])[:, 1])\n\n# 3. Thuật toán ACO tối ưu\nn_features = X_train_encoded.shape[1]\npheromone = np.ones(n_features)\nbest_auc = 0\nbest_mask = np.ones(n_features, dtype=bool)\nhistory_auc = []\n\nprint(\"🐜 Đàn kiến bắt đầu hành trình (Tối ưu hóa)...\")\n\nfor iteration in range(20): # Giảm xuống 20 vòng để chạy nhanh hơn\n    # Tạo các mask ngẫu nhiên, đảm bảo không có mask nào rỗng\n    masks = []\n    for _ in range(10): # Giảm số kiến xuống 10 để tránh quá tải CPU\n        m = (np.random.rand(n_features) < (pheromone / pheromone.sum()))\n        if np.sum(m) == 0: m[np.random.randint(0, n_features)] = True\n        masks.append(m)\n    \n    # Chạy song song\n    scores = Parallel(n_jobs=-1)(delayed(evaluate_features)(m, X_train_encoded, y_train, X_val_encoded, y_val) for m in masks)\n    \n    # Cập nhật kết quả tốt nhất\n    for i, score in enumerate(scores):\n        if score > best_auc:\n            best_auc = score\n            best_mask = masks[i]\n            \n    history_auc.append(best_auc)\n    pheromone *= 0.9 # Bay hơi\n    pheromone[best_mask] += 0.1 # Củng cố pheromone\n    \n    print(f\"Vòng {iteration+1}: Best AUC = {best_auc:.4f}\")\n    gc.collect() # Dọn dẹp RAM sau mỗi vòng lặp\n\nprint(f\"\\n✅ Hoàn tất! Best Mask: {best_mask.astype(int)}\")\nnp.save('best_mask.npy', best_mask)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T12:13:42.504876Z","iopub.execute_input":"2026-06-22T12:13:42.505143Z","iopub.status.idle":"2026-06-22T15:26:35.158427Z","shell.execute_reply.started":"2026-06-22T12:13:42.505119Z","shell.execute_reply":"2026-06-22T15:26:35.157555Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(8, 5))\nplt.plot(history_auc, marker='o', linestyle='-', color='orange')\nplt.title('Quá trình tối ưu hóa của Đàn Kiến')\nplt.xlabel('Vòng lặp (Iteration)')\nplt.ylabel('AUC Score')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T15:26:35.160708Z","iopub.execute_input":"2026-06-22T15:26:35.161032Z","iopub.status.idle":"2026-06-22T15:26:35.450749Z","shell.execute_reply.started":"2026-06-22T15:26:35.161003Z","shell.execute_reply":"2026-06-22T15:26:35.449541Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import layers, models\n\n# 1. Lọc dữ liệu theo best_mask\nbest_mask = np.load('best_mask.npy').astype(bool)\nX_train_final = X_train_encoded[:, best_mask]\nX_val_final = X_val_encoded[:, best_mask]\n\n# 2. Xây dựng mô hình CNN 1D\nmodel = models.Sequential([\n    # Input cần định dạng (số_feature, 1) cho CNN 1D\n    layers.Input(shape=(X_train_final.shape[1], 1)),\n    layers.Conv1D(32, kernel_size=2, activation='relu'),\n    layers.MaxPooling1D(pool_size=1),\n    layers.Flatten(),\n    layers.Dense(16, activation='relu'),\n    layers.Dense(1, activation='sigmoid') # Phân loại 0 hoặc 1\n])\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['AUC'])\n\n# 3. Huấn luyện\n# Reshape cho CNN 1D: (samples, time_steps, features)\nX_train_cnn = X_train_final.reshape(X_train_final.shape[0], X_train_final.shape[1], 1)\nX_val_cnn = X_val_final.reshape(X_val_final.shape[0], X_val_final.shape[1], 1)\n\nhistory = model.fit(X_train_cnn, y_train, epochs=10, batch_size=128, \n                    validation_data=(X_val_cnn, y_val))\n\n# 4. Lưu mô hình\nmodel.save('cnn_fraud_detection.keras')\nprint(\"✅ Mô hình CNN 1D đã huấn luyện và lưu thành công!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T15:26:35.452729Z","iopub.execute_input":"2026-06-22T15:26:35.453567Z","iopub.status.idle":"2026-06-22T15:53:17.408495Z","shell.execute_reply.started":"2026-06-22T15:26:35.453533Z","shell.execute_reply":"2026-06-22T15:53:17.406300Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_auc_cnn = history.history['val_AUC'][-1] \nprint(f\"✅ AUC cuối cùng của CNN 1D là: {final_auc_cnn:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T16:02:23.318823Z","iopub.execute_input":"2026-06-22T16:02:23.319159Z","iopub.status.idle":"2026-06-22T16:02:23.324902Z","shell.execute_reply.started":"2026-06-22T16:02:23.319131Z","shell.execute_reply":"2026-06-22T16:02:23.323742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.plot(history.history['loss'], label='Training Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Biểu đồ Loss của mô hình CNN 1D')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T16:04:41.469950Z","iopub.execute_input":"2026-06-22T16:04:41.470807Z","iopub.status.idle":"2026-06-22T16:04:41.641034Z","shell.execute_reply.started":"2026-06-22T16:04:41.470743Z","shell.execute_reply":"2026-06-22T16:04:41.640011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nplt.plot(history.history['AUC'], label='Training AUC', color='green')\nplt.plot(history.history['val_AUC'], label='Validation AUC', color='red', linestyle='--')\nplt.title('Biểu đồ AUC của mô hình CNN 1D qua các Epoch')\nplt.xlabel('Epochs')\nplt.ylabel('AUC')\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T16:04:46.120474Z","iopub.execute_input":"2026-06-22T16:04:46.120807Z","iopub.status.idle":"2026-06-22T16:04:46.304213Z","shell.execute_reply.started":"2026-06-22T16:04:46.120778Z","shell.execute_reply":"2026-06-22T16:04:46.303168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#============================================\n# CELL: BIỂU ĐỒ SO SÁNH HIỆU SUẤT (FINAL REPORT)\n# ============================================\nfrom sklearn.metrics import roc_auc_score\n\n# 1. Tính AUC của CNN 1D (đã train xong)\ny_pred_cnn = model.predict(X_val_cnn)\nauc_cnn = roc_auc_score(y_val, y_pred_cnn)\n\n# 2. Giả sử bạn đã có AUC của các mô hình trước (điền vào đây)\nmodels = ['Random Forest', 'XGBoost', 'LightGBM', 'CNN 1D (Proposed)']\nauc_scores = [0.9559, 0.9666, 0.9155, final_auc_cnn] # Bạn thay số thật vào nhé\n\n# 3. Vẽ biểu đồ\nplt.figure(figsize=(10, 6))\nplt.bar(models, auc_scores, color=['#999999', '#999999', '#999999', '#ff9900'])\nplt.title('So sánh AUC giữa các mô hình Baseline và CNN 1D', fontsize=15)\nplt.ylabel('AUC Score')\nplt.ylim(min(auc_scores)-0.05, 1.0) # Zoom vào khu vực quan trọng\nfor i, v in enumerate(auc_scores):\n    plt.text(i, v + 0.005, f\"{v:.4f}\", ha='center', fontweight='bold')\nplt.grid(axis='y', linestyle='--', alpha=0.5)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-22T16:02:55.044117Z","iopub.execute_input":"2026-06-22T16:02:55.044502Z","iopub.status.idle":"2026-06-22T16:04:26.547260Z","shell.execute_reply.started":"2026-06-22T16:02:55.044427Z","shell.execute_reply":"2026-06-22T16:04:26.545944Z"}},"outputs":[],"execution_count":null}]}