{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"datasetVersion","sourceId":6067645,"datasetId":3472723,"databundleVersionId":6145994}],"dockerImageVersionId":30096,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Import libraries","metadata":{}},{"cell_type":"code","source":"! pip install simdkalman","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:24:50.910036Z","iopub.execute_input":"2026-05-08T16:24:50.910490Z","iopub.status.idle":"2026-05-08T16:24:57.231148Z","shell.execute_reply.started":"2026-05-08T16:24:50.910455Z","shell.execute_reply":"2026-05-08T16:24:57.229966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport simdkalman\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2026-05-08T16:24:57.233215Z","iopub.execute_input":"2026-05-08T16:24:57.233675Z","iopub.status.idle":"2026-05-08T16:24:57.238736Z","shell.execute_reply.started":"2026-05-08T16:24:57.233616Z","shell.execute_reply":"2026-05-08T16:24:57.237582Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Read ACB dataset from the large Stock Prices & Volume VN30 Index Vietnam","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/datasets/thangtranquang/stock-vn30-vietnam/ACB.csv')\ndf['TradingDate'] = pd.to_datetime(df['TradingDate'], dayfirst=True)\ndf = df.sort_values('TradingDate').reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2026-05-08T16:24:57.240880Z","iopub.execute_input":"2026-05-08T16:24:57.241327Z","iopub.status.idle":"2026-05-08T16:24:57.271154Z","shell.execute_reply.started":"2026-05-08T16:24:57.241283Z","shell.execute_reply":"2026-05-08T16:24:57.270230Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reject outlier","metadata":{}},{"cell_type":"code","source":"# Lọc Outlier (Giống bước Reject Outlier trong file GPS)\n# Nếu giá nhảy vọt quá 5% trong 1 phiên (nhiễu), ta coi là bất thường\ndf['price_diff_pct'] = df['Close'].pct_change().abs()\nthreshold_pct = 0.05 \ndf.loc[df['price_diff_pct'] > threshold_pct, 'Close'] = np.nan\n\n# Lấp đầy các khoảng trống NaN bằng nội suy tuyến tính \ndf['Close'] = df['Close'].interpolate(method='linear')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:24:57.272674Z","iopub.execute_input":"2026-05-08T16:24:57.272949Z","iopub.status.idle":"2026-05-08T16:24:57.284188Z","shell.execute_reply.started":"2026-05-08T16:24:57.272916Z","shell.execute_reply":"2026-05-08T16:24:57.282919Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Make lerp Data","metadata":{}},{"cell_type":"code","source":"def make_lerp_data(df):\n    '''\n    Tạo dữ liệu nội suy cho các ngày bị thiếu trong chuỗi thời gian ACB.\n    '''\n    # 1. Tạo ra trục thời gian đầy đủ từ ngày bắt đầu đến ngày kết thúc\n    full_range = pd.date_range(start=df['TradingDate'].min(), end=df['TradingDate'].max(), freq='D')\n    full_df = pd.DataFrame({'TradingDate': full_range})\n    \n    # 2. Gộp dữ liệu hiện có vào trục thời gian đầy đủ\n    # Những ngày không có giao dịch sẽ mang giá trị NaN\n    df_merged = full_df.merge(df, on='TradingDate', how='left')\n    \n    # 3. Thực hiện nội suy tuyến tính (Linear Interpolation) để lấp đầy các ngày NaN\n    # Điều này giúp Kalman Filter có dữ liệu liên tục để tính toán vận tốc/xu hướng\n    df_merged['Close'] = df_merged['Close'].interpolate(method='linear')\n    \n    # Nếu có các cột khác như Open, High, Low, bạn cũng có thể nội suy tương tự\n    return df_merged","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:24:57.285519Z","iopub.execute_input":"2026-05-08T16:24:57.285912Z","iopub.status.idle":"2026-05-08T16:24:57.294823Z","shell.execute_reply.started":"2026-05-08T16:24:57.285881Z","shell.execute_reply":"2026-05-08T16:24:57.293669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = make_lerp_data(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:24:57.296422Z","iopub.execute_input":"2026-05-08T16:24:57.296768Z","iopub.status.idle":"2026-05-08T16:24:57.323534Z","shell.execute_reply.started":"2026-05-08T16:24:57.296710Z","shell.execute_reply":"2026-05-08T16:24:57.322248Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Chia dữ liệu train/test","metadata":{}},{"cell_type":"code","source":"split_idx = int(len(df) * 0.8)\ntrain_df = df.iloc[:split_idx].copy()\ntest_df = df.iloc[split_idx:].copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:24:57.324929Z","iopub.execute_input":"2026-05-08T16:24:57.325455Z","iopub.status.idle":"2026-05-08T16:24:57.331548Z","shell.execute_reply.started":"2026-05-08T16:24:57.325421Z","shell.execute_reply":"2026-05-08T16:24:57.330514Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Setup Kalman Filtering","metadata":{}},{"cell_type":"code","source":"# Trạng thái: [Giá, Xu hướng]\nT = 1.0\nstate_transition = np.array([[1, T], \n                             [0, 1]])\nobservation_model = np.array([[1, 0]])\n\n# Thiết lập mức độ tin tưởng (Tuning)\n# Observation noise (R): Nhiễu thị trường \nprocess_noise = np.diag([1e-3, 1e-4]) \nobservation_noise = np.array([[1000**2]]) \n\nkf = simdkalman.KalmanFilter(\n    state_transition = state_transition,\n    process_noise = process_noise,\n    observation_model = observation_model,\n    observation_noise = observation_noise\n)\n\ndef apply_kalman_smoothing(series):\n    data = series.values.reshape(1, len(series), 1)\n    smoothed = kf.smooth(data)\n    return smoothed.states.mean[0, :, 0]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:24:57.333654Z","iopub.execute_input":"2026-05-08T16:24:57.333925Z","iopub.status.idle":"2026-05-08T16:24:57.348157Z","shell.execute_reply.started":"2026-05-08T16:24:57.333899Z","shell.execute_reply":"2026-05-08T16:24:57.346878Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Filter and Evaluate","metadata":{}},{"cell_type":"code","source":"# 1. Lọc trên tập Test\ntest_df['Close_Kalman'] = apply_kalman_smoothing(test_df['Close'])\n\n# 2. Tính điểm Score\ndef evaluate_score(df_result):\n    df_result['err'] = (df_result['Close'] - df_result['Close_Kalman']).abs()\n    \n    p50 = np.percentile(df_result['err'], 50)\n    p95 = np.percentile(df_result['err'], 95)\n    score = (p50 + p95) / 2\n    return score, p50, p95\n\nfinal_score, p50, p95 = evaluate_score(test_df)\n\nprint(f\"--- ĐÁNH GIÁ TRÊN TẬP TEST ---\")\nprint(f\"Điểm sai lệch tổng hợp (Score): {final_score:.2f} VNĐ\")\nprint(f\"Sai số trung vị (P50): {p50:.2f} VNĐ\")\nprint(f\"Sai số cực đại (P95): {p95:.2f} VNĐ\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:24:57.349971Z","iopub.execute_input":"2026-05-08T16:24:57.350406Z","iopub.status.idle":"2026-05-08T16:24:57.593151Z","shell.execute_reply.started":"2026-05-08T16:24:57.350363Z","shell.execute_reply":"2026-05-08T16:24:57.591941Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Make submission","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 7))\nplt.plot(test_df['TradingDate'], test_df['Close'], label='Giá thực tế (Test)', color='blue', alpha=0.4)\nplt.plot(test_df['TradingDate'], test_df['Close_Kalman'], label='Giá sau lọc Kalman', color='red', linewidth=2)\nplt.title('Kết quả lọc Kalman Filter trên tập Test (Mã ACB)')\nplt.legend()\nplt.show()\n\n# Lưu kết quả tập Test ra file .csv \ntest_df[['TradingDate', 'Close', 'Close_Kalman']].to_csv('ACB_Submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-08T16:24:57.594572Z","iopub.execute_input":"2026-05-08T16:24:57.594899Z","iopub.status.idle":"2026-05-08T16:24:57.861076Z","shell.execute_reply.started":"2026-05-08T16:24:57.594868Z","shell.execute_reply":"2026-05-08T16:24:57.860216Z"}},"outputs":[],"execution_count":null}]}