{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4117,"databundleVersionId":46665,"isSourceIdPinned":false}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-04-11T15:30:49.435490Z","iopub.execute_input":"2026-04-11T15:30:49.435951Z","iopub.status.idle":"2026-04-11T15:30:49.791853Z","shell.execute_reply.started":"2026-04-11T15:30:49.435917Z","shell.execute_reply":"2026-04-11T15:30:49.790825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q py7zr tqdm\nprint('OK')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T15:30:49.793818Z","iopub.execute_input":"2026-04-11T15:30:49.794224Z","iopub.status.idle":"2026-04-11T15:30:57.100191Z","shell.execute_reply.started":"2026-04-11T15:30:49.794193Z","shell.execute_reply":"2026-04-11T15:30:57.099136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, csv, time, logging, shutil\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom pathlib import Path\nfrom datetime import datetime\nfrom tqdm.notebook import tqdm\nimport py7zr\n\nINPUT_DIR     = Path('/kaggle/input/competitions/malware-classification')\nWORKING_DIR   = Path('/kaggle/working')\nIMAGE_OUT_DIR = WORKING_DIR / 'malware_images'\nINDEX_CSV     = WORKING_DIR / 'malware_index.csv'\nLOG_FILE      = WORKING_DIR / 'preprocessing_log.txt'\nTEMP_DIR      = WORKING_DIR / '_tmp'\nTRAIN_7Z      = INPUT_DIR / 'train.7z'\nTEST_7Z       = INPUT_DIR / 'test.7z'\nLABELS_CSV    = INPUT_DIR / 'trainLabels.csv'\n\nIMG_SIZE      = 224\nBYTES_PER_ROW = 256\nBATCH_SIZE    = 200\n\nFAMILY_MAP = {\n    1:'Ramnit', 2:'Lollipop', 3:'Kelihos_ver3',\n    4:'Vundo',  5:'Simda',    6:'Tracur',\n    7:'Kelihos_ver1', 8:'Obfuscator_ACY', 9:'Gatak',\n}\nCSV_COLUMNS = [\n    'file_id','split','label_id','family_name',\n    'status','image_path','orig_height','orig_width',\n    'error_msg','processed_at',\n]\nIMAGE_OUT_DIR.mkdir(parents=True, exist_ok=True)\nlogging.basicConfig(filename=str(LOG_FILE), level=logging.INFO,\n    format='%(asctime)s|%(levelname)s|%(message)s')\nlogger = logging.getLogger(__name__)\n\ndef disk_free_gb(): return shutil.disk_usage('/kaggle/working').free / 1024**3\ndef disk_used_gb(): return shutil.disk_usage('/kaggle/working').used / 1024**3\nprint('Config OK  |  Disk free:', round(disk_free_gb(),2), 'GB')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T15:30:57.101616Z","iopub.execute_input":"2026-04-11T15:30:57.101939Z","iopub.status.idle":"2026-04-11T15:30:57.557118Z","shell.execute_reply.started":"2026-04-11T15:30:57.101904Z","shell.execute_reply":"2026-04-11T15:30:57.556215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Xoa toan bo anh PNG cu (duoc tao bang rb - sai noi dung)\nif IMAGE_OUT_DIR.exists():\n    count = len(list(IMAGE_OUT_DIR.rglob('*.png')))\n    print(f'Dang xoa {count:,} file PNG cu (sai noi dung)...')\n    shutil.rmtree(IMAGE_OUT_DIR)\n    IMAGE_OUT_DIR.mkdir(parents=True, exist_ok=True)\n    print('[OK] Da xoa het anh cu')\n\n# Reset tat ca record 'done' -> 'pending' trong index\nif INDEX_CSV.exists():\n    df = pd.read_csv(INDEX_CSV).drop_duplicates(subset=['file_id'], keep='last')\n    print(f'Index cu: {len(df):,} records, {(df.status==\"done\").sum():,} done')\n    df['status'] = 'pending'\n    df['image_path'] = ''\n    df['orig_height'] = ''\n    df['orig_width'] = ''\n    df['error_msg'] = ''\n    df['processed_at'] = ''\n    df.to_csv(INDEX_CSV, index=False)\n    print(f'[OK] Reset {len(df):,} records -> pending')\n\nprint(f'Disk free: {disk_free_gb():.2f} GB')\nprint('San sang chay lai Cell 6 va Cell 7.')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T15:30:57.558803Z","iopub.execute_input":"2026-04-11T15:30:57.559407Z","iopub.status.idle":"2026-04-11T15:30:57.568198Z","shell.execute_reply.started":"2026-04-11T15:30:57.559374Z","shell.execute_reply":"2026-04-11T15:30:57.567203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Tim 1 file PNG bat ky da luu\npngs = list(IMAGE_OUT_DIR.rglob('*.png'))\nprint(f'PNG tren disk: {len(pngs):,}')\nif pngs:\n    p = pngs[0]\n    img = Image.open(p)\n    arr = np.array(img)\n    print(f'File: {p.name}')\n    print(f'Shape: {arr.shape}  dtype: {arr.dtype}')\n    print(f'min={arr.min()}  max={arr.max()}  mean={arr.mean():.1f}')\n    print('OK neu max > 100 (anh co texture thuc su, khong phai chi ASCII chars)')\n    # Histogram nhanh\n    unique, counts = np.unique(arr, return_counts=True)\n    print(f'So gia tri pixel khac nhau: {len(unique)} (tot: >50, xau: <30)')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T15:30:57.570494Z","iopub.execute_input":"2026-04-11T15:30:57.570903Z","iopub.status.idle":"2026-04-11T15:30:57.593151Z","shell.execute_reply.started":"2026-04-11T15:30:57.570871Z","shell.execute_reply":"2026-04-11T15:30:57.592207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_index():\n    if not INDEX_CSV.exists():\n        print('[i] Chua co checkpoint')\n        return {}\n    index = {}\n    with open(INDEX_CSV,'r',newline='',encoding='utf-8') as f:\n        for row in csv.DictReader(f):\n            index[row['file_id']] = row\n    done  = sum(1 for r in index.values() if r['status']=='done')\n    error = sum(1 for r in index.values() if r['status']=='error')\n    pend  = sum(1 for r in index.values() if r['status']=='pending')\n    print(f'[OK] Index: {len(index):,} | done={done:,} error={error} pending={pend:,}')\n    return index\n\ndef append_row(row):\n    need_header = not INDEX_CSV.exists()\n    with open(INDEX_CSV,'a',newline='',encoding='utf-8') as f:\n        w = csv.DictWriter(f, fieldnames=CSV_COLUMNS)\n        if need_header: w.writeheader()\n        w.writerow({k: row.get(k,'') for k in CSV_COLUMNS})\n\ndef print_summary(index):\n    total = len(index)\n    done  = sum(1 for r in index.values() if r['status']=='done')\n    error = sum(1 for r in index.values() if r['status']=='error')\n    pend  = sum(1 for r in index.values() if r['status']=='pending')\n    pct   = done/total*100 if total>0 else 0\n    print(f'  Total={total:,}  Done={done:,}({pct:.1f}%)  Error={error}  Pending={pend:,}')\n\ndef bytes_file_to_array(bytes_path):\n    \"\"\"\n    BIG-2015 .bytes = HEX TEXT format:\n        00401000 E8 0B 00 00 00 E9 16 00 ...\n        ^ADDRESS  ^--- gia tri hex (0-FF) ---^\n\n    Parse dung:\n    - Moi dong: token dau la dia chi (8 ky tu hex) -> BO QUA\n    - Cac token con lai: gia tri byte HEX 00-FF, hoac '??' -> 0\n    \"\"\"\n    raw = []\n    with open(bytes_path, 'r', errors='replace') as f:\n        for line in f:\n            tokens = line.split()\n            if not tokens:\n                continue\n            # Token dau tien la dia chi (vi du: '00401000') -> skip\n            # Kiem tra: dia chi co dung 8 ky tu hex khong\n            start = 0\n            if len(tokens[0]) == 8:\n                try:\n                    int(tokens[0], 16)  # la hex -> la dia chi -> skip\n                    start = 1\n                except ValueError:\n                    pass  # khong phai hex -> giu nguyen\n            for tok in tokens[start:]:\n                if tok == '??':\n                    raw.append(0)\n                elif len(tok) <= 2:\n                    try:\n                        raw.append(int(tok, 16))\n                    except ValueError:\n                        raw.append(0)\n                # Neu tok dai hon 2 ky tu -> khong phai byte -> bo qua\n\n    if not raw:\n        raise ValueError('Khong co du lieu byte hop le')\n\n    arr = np.array(raw, dtype=np.uint8)\n    rem = arr.size % BYTES_PER_ROW\n    if rem:\n        arr = np.concatenate([arr, np.zeros(BYTES_PER_ROW - rem, dtype=np.uint8)])\n\n    orig_h = arr.size // BYTES_PER_ROW\n    arr2d  = arr.reshape(orig_h, BYTES_PER_ROW)\n    img    = Image.fromarray(arr2d).resize((IMG_SIZE, IMG_SIZE), Image.NEAREST)\n    return np.array(img, dtype=np.uint8), orig_h, BYTES_PER_ROW\n\ndef save_png(arr, file_id, split, family):\n    out_dir = IMAGE_OUT_DIR / split / family\n    out_dir.mkdir(parents=True, exist_ok=True)\n    path = out_dir / f'{file_id}.png'\n    Image.fromarray(arr).save(str(path))\n    return str(path)\n\ndef list_7z_bytes(archive_path):\n    with py7zr.SevenZipFile(str(archive_path), mode='r') as z:\n        return [n for n in z.getnames() if n.endswith('.bytes')]\n\ndef find_bytes_file(temp_dir, archive_name):\n    p = temp_dir / archive_name\n    if p.exists(): return p\n    fname = Path(archive_name).name\n    p2 = temp_dir / fname\n    if p2.exists(): return p2\n    matches = list(temp_dir.rglob(fname))\n    if matches: return matches[0]\n    raise FileNotFoundError(f'Khong tim thay {fname}')\n\ndef process_batch(archive_path, batch_names, index, split, pbar):\n    TEMP_DIR.mkdir(parents=True, exist_ok=True)\n    done_n = error_n = 0\n    with py7zr.SevenZipFile(str(archive_path), mode='r') as z:\n        z.extract(targets=batch_names, path=str(TEMP_DIR))\n    for name in batch_names:\n        fid    = Path(name).stem\n        record = index.get(fid, {})\n        bytes_path = None\n        try:\n            bytes_path = find_bytes_file(TEMP_DIR, name)\n            arr, orig_h, orig_w = bytes_file_to_array(bytes_path)\n            family   = record.get('family_name', 'Unknown')\n            img_path = save_png(arr, fid, split, family)\n            bytes_path.unlink()  # chi xoa .bytes, KHONG xoa PNG\n            bytes_path = None\n            row = {\n                'file_id':fid, 'split':split,\n                'label_id':record.get('label_id',-1),\n                'family_name':family,\n                'status':'done', 'image_path':img_path,\n                'orig_height':orig_h, 'orig_width':orig_w,\n                'error_msg':'', 'processed_at':datetime.now().isoformat(),\n            }\n            index[fid] = row\n            append_row(row)\n            done_n += 1\n        except Exception as e:\n            if bytes_path and bytes_path.exists(): bytes_path.unlink()\n            msg = str(e)[:300]\n            row = {\n                'file_id':fid, 'split':split,\n                'label_id':record.get('label_id',-1),\n                'family_name':record.get('family_name','Unknown'),\n                'status':'error', 'image_path':'',\n                'orig_height':'', 'orig_width':'',\n                'error_msg':msg, 'processed_at':datetime.now().isoformat(),\n            }\n            index[fid] = row\n            append_row(row)\n            error_n += 1\n            logger.error(f'{fid} | {msg}')\n        pbar.update(1)\n    shutil.rmtree(TEMP_DIR, ignore_errors=True)\n    return done_n, error_n\n\nprint('Helpers OK')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T15:30:57.594506Z","iopub.execute_input":"2026-04-11T15:30:57.595326Z","iopub.status.idle":"2026-04-11T15:30:57.624295Z","shell.execute_reply.started":"2026-04-11T15:30:57.595290Z","shell.execute_reply":"2026-04-11T15:30:57.623367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SPLIT = 'train'\nARCHIVE = TRAIN_7Z\nprint('='*55)\nprint(f'  {ARCHIVE.name}  |  {datetime.now().strftime(\"%H:%M:%S\")}')\nprint(f'  Disk free: {disk_free_gb():.2f} GB')\nprint('='*55)\n\nlabel_map = dict(zip(\n    pd.read_csv(LABELS_CSV)['Id'].astype(str),\n    pd.read_csv(LABELS_CSV)['Class'].astype(int)\n))\nindex = load_index()\nall_names = list_7z_bytes(ARCHIVE)\nprint(f'[OK] {len(all_names):,} .bytes files')\n\nnew_count = 0\nfor name in all_names:\n    fid = Path(name).stem\n    if fid not in index:\n        label_id    = label_map.get(fid, -1)\n        family_name = FAMILY_MAP.get(label_id, 'Unknown')\n        record = {'file_id':fid,'split':SPLIT,'label_id':label_id,\n            'family_name':family_name,'status':'pending',\n            'image_path':'','orig_height':'','orig_width':'',\n            'error_msg':'','processed_at':''}\n        index[fid] = record\n        append_row(record)\n        new_count += 1\nif new_count: print(f'[OK] Them {new_count:,} file moi')\n\npending = [n for n in all_names if index.get(Path(n).stem,{}).get('status')=='pending']\nprint(f'[OK] Pending: {len(pending):,}  |  Disk free: {disk_free_gb():.2f} GB')\n\nif not pending:\n    print('[DONE] Train xong!')\n    print_summary(index)\nelse:\n    # Test 1 file truoc\n    print('\\n[TEST] Thu convert 1 file...')\n    TEMP_DIR.mkdir(parents=True, exist_ok=True)\n    try:\n        test_name = pending[0]\n        with py7zr.SevenZipFile(str(ARCHIVE), mode='r') as z:\n            z.extract(targets=[test_name], path=str(TEMP_DIR))\n        test_path = find_bytes_file(TEMP_DIR, test_name)\n        arr, h, w = bytes_file_to_array(test_path)\n        print(f'  shape={arr.shape}  min={arr.min()}  max={arr.max()}  mean={arr.mean():.1f}')\n        if arr.max() < 10:\n            raise ValueError('max pixel qua thap - anh co the bi sai!')\n        test_path.unlink()\n        shutil.rmtree(TEMP_DIR, ignore_errors=True)\n        print('[OK] Test passed')\n    except Exception as e:\n        shutil.rmtree(TEMP_DIR, ignore_errors=True)\n        raise RuntimeError(f'Test that bai: {e}') from e\n\n    batches = [pending[i:i+BATCH_SIZE] for i in range(0, len(pending), BATCH_SIZE)]\n    total_done = total_error = 0\n    t0 = time.time()\n    pbar = tqdm(total=len(pending), desc='train', unit='file',\n        bar_format='{l_bar}{bar}| {n_fmt}/{total_fmt} [{elapsed}<{remaining}, {rate_fmt}]')\n    for b_idx, batch in enumerate(batches):\n        tqdm.write(f'  Batch {b_idx+1:>3}/{len(batches)} | free={disk_free_gb():.2f}GB | done={total_done:,} err={total_error}')\n        if disk_free_gb() < 0.5:\n            tqdm.write('[WARN] Disk < 0.5GB -- dung!')\n            break\n        done_n, error_n = process_batch(ARCHIVE, batch, index, SPLIT, pbar)\n        total_done += done_n; total_error += error_n\n    pbar.close()\n    print(f'\\n[OK] {(time.time()-t0)/60:.1f} min | done={total_done:,} err={total_error}')\n    print_summary(index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-11T15:30:57.625323Z","iopub.execute_input":"2026-04-11T15:30:57.625645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SPLIT = 'test'\nARCHIVE = TEST_7Z\nprint('='*55)\nprint(f'  {ARCHIVE.name}  |  {datetime.now().strftime(\"%H:%M:%S\")}')\nprint(f'  Disk free: {disk_free_gb():.2f} GB')\nprint('='*55)\n\nlabel_map = dict(zip(\n    pd.read_csv(LABELS_CSV)['Id'].astype(str),\n    pd.read_csv(LABELS_CSV)['Class'].astype(int)\n))\nindex = load_index()\nall_names = list_7z_bytes(ARCHIVE)\nprint(f'[OK] {len(all_names):,} .bytes files')\n\nnew_count = 0\nfor name in all_names:\n    fid = Path(name).stem\n    if fid not in index:\n        record = {'file_id':fid,'split':SPLIT,'label_id':-1,\n            'family_name':'Unknown','status':'pending',\n            'image_path':'','orig_height':'','orig_width':'',\n            'error_msg':'','processed_at':''}\n        index[fid] = record\n        append_row(record)\n        new_count += 1\nif new_count: print(f'[OK] Them {new_count:,} file moi')\n\npending = [n for n in all_names if index.get(Path(n).stem,{}).get('status')=='pending']\nprint(f'[OK] Pending: {len(pending):,}')\n\nif not pending:\n    print('[DONE] Test xong!')\n    print_summary(index)\nelse:\n    batches = [pending[i:i+BATCH_SIZE] for i in range(0, len(pending), BATCH_SIZE)]\n    total_done = total_error = 0\n    t0 = time.time()\n    pbar = tqdm(total=len(pending), desc='test', unit='file',\n        bar_format='{l_bar}{bar}| {n_fmt}/{total_fmt} [{elapsed}<{remaining}, {rate_fmt}]')\n    for b_idx, batch in enumerate(batches):\n        tqdm.write(f'  Batch {b_idx+1:>3}/{len(batches)} | free={disk_free_gb():.2f}GB | done={total_done:,} err={total_error}')\n        if disk_free_gb() < 0.5:\n            tqdm.write('[WARN] Disk < 0.5GB -- dung!')\n            break\n        done_n, error_n = process_batch(ARCHIVE, batch, index, SPLIT, pbar)\n        total_done += done_n; total_error += error_n\n    pbar.close()\n    print(f'\\n[OK] {(time.time()-t0)/60:.1f} min | done={total_done:,} err={total_error}')\n    print_summary(index)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not INDEX_CSV.exists():\n    print('[!] Chua co index.')\nelse:\n    df = pd.read_csv(INDEX_CSV).drop_duplicates(subset=['file_id'], keep='last')\n    png_count = len(list(IMAGE_OUT_DIR.rglob('*.png')))\n    print(f'Index: {len(df):,} | PNG tren disk: {png_count:,} | Disk free: {disk_free_gb():.2f} GB')\n    print('\\n-- Status --')\n    print(df['status'].value_counts().to_string())\n    print('\\n-- Train done theo family --')\n    dt = df[(df['status']=='done')&(df['split']=='train')]\n    print(dt['family_name'].value_counts().sort_index().to_string() if len(dt) else '  (chua co)')\n    print('\\n-- Split x Status --')\n    print(df.groupby(['split','status']).size().to_string())","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}