# -*- coding: utf-8 -*-
"""从 .smart 项目文件解压并提取完整文本（ASCII + GBK 中文）"""
import os, zlib, sys

SRC = r'D:\你的项目目录\技术资料\你的源程序目录'
DST = r'D:\你的工作目录\05_工具脚本\_extract'
os.makedirs(DST, exist_ok=True)


def extract_text(dec: bytes) -> str:
    """把解压后的字节流里所有可读内容（ASCII + GBK 中文）抽成文本，保留换行"""
    out = []
    buf = bytearray()
    i = 0
    n = len(dec)
    while i < n:
        c = dec[i]
        if c in (0x0d, 0x0a, 0x09):
            buf.append(c)
            i += 1
        elif 0x20 <= c <= 0x7e:
            buf.append(c)
            i += 1
        elif c >= 0x81 and i + 1 < n:
            b2 = dec[i + 1]
            if (0x40 <= b2 <= 0x7e) or (0x80 <= b2 <= 0xfe):
                try:
                    s = bytes([c, b2]).decode('gbk')
                    buf += s.encode('utf-8')
                    i += 2
                    continue
                except Exception:
                    pass
            if buf:
                out.append(bytes(buf).decode('utf-8', 'replace'))
                buf = bytearray()
            i += 1
        else:
            if buf:
                out.append(bytes(buf).decode('utf-8', 'replace'))
                buf = bytearray()
            i += 1
    if buf:
        out.append(bytes(buf).decode('utf-8', 'replace'))

    # 只保留长度>=3的连续文本段，压掉噪声
    res = []
    for s in out:
        s = s.strip('\x00')
        if len(s.strip()) >= 3:
            res.append(s)
    return '\n'.join(res)


report = []
for f in sorted(os.listdir(SRC)):
    p = os.path.join(SRC, f)
    b = open(p, 'rb').read()
    # 找 zlib 流
    best = None
    for i in range(len(b) - 2):
        if b[i] == 0x78 and b[i + 1] in (0x01, 0x5e, 0x9c, 0xda):
            do = zlib.decompressobj()
            try:
                dec = do.decompress(b[i:])
            except Exception:
                continue
            if best is None or len(dec) > len(best[1]):
                best = (i, dec)
            break
    if best is None:
        report.append('%s: NO zlib stream found' % f)
        continue
    off, dec = best
    txt = extract_text(dec)
    outname = os.path.splitext(f)[0].replace(' ', '_') + '.txt'
    with open(os.path.join(DST, outname), 'w', encoding='utf-8') as fh:
        fh.write(txt)
    report.append('%s -> %s | zlib@%d, %d bytes raw, %d chars text, %d lines'
                  % (f, outname, off, len(dec), len(txt), txt.count('\n') + 1))

print('\n'.join(report))
