python脚本

#!/usr/bin/env python3
"""向 lv_font_conv 生成的 LVGL 字体 C 文件增量添加字符。

用法(--font 输入字体、--out 输出文件为必填, 相对/绝对路径均可):
  python3 add_font_chars.py --font simhei.ttf --out lvgl/src/font/songti_20.c 新
  python3 add_font_chars.py --font /abs/path/a.ttf --out /abs/path/b.c 新 操 满   # 一次添加多个字
  python3 add_font_chars.py --font simhei.ttf --out out.c --gen generated.c 新     # 用官网生成的文件代替 lv_font_conv
--gen 是数据源替换选项:跳过本地 lv_font_conv,改用你自己准备的生成文件来提供新字的位图数据。

说明: 已存在于字体中的字符会自动跳过并打印日志, 不会报错。
     超出当前 cmap 范围的字符会自动扩展 range_start/range_length(中文字符的 rcp 必然在
     uint16 范围内, 无需新增 cmap)。
原理:
  1. 用 lv_font_conv 只生成新增字符的字形数据(与目标文件同 bpp/size/无压缩, 格式保证一致)
  2. 解析出每个新字的位图字节和字形参数(adv_w/box_w/box_h/ofs_x/ofs_y)
  3. 按 LVGL fmt_txt 规则拼入目标文件:
     - glyph_bitmap: 字节追加到数组末尾, bitmap_index = 当前总字节数
     - glyph_dsc:   按 glyph_id (= glyph_id_start + 在 unicode_list 中的下标)插入, 后续条目顺延
     - unicode_list: 保持升序插入(LVGL 二分查找要求), list_length 同步加
     - cmap:         range_start/range_length 按需自动扩展
  4. 自校验: 升序性 / 位图区间无重叠越界 / 字节数一致 / 字形映射正确
"""

import argparse
import bisect
import math
import os
import re
import shutil
import subprocess
import sys
import tempfile

TOKEN_RE = re.compile(r'0x[0-9a-fA-F]+')
DSC_RE = re.compile(
    r'\.bitmap_index = (\d+),\s*\.adv_w = (\d+),\s*\.box_w = (\d+),\s*\.box_h = (\d+),\s*\.ofs_x = (-?\d+),\s*\.ofs_y = (-?\d+)')
GLYPH_COMMENT_RE = re.compile(r'/\* U\+([0-9A-Fa-f]{1,6}) "([^"]*)" \*/')
CMAP_FIRST_LINE_RE = re.compile(
    r'\.range_start = (\d+),\s*\.range_length = (\d+),\s*\.glyph_id_start = (\d+),')
LISTLEN_RE = re.compile(r'\.list_length = (\d+)')


def die(msg):
    print("错误: " + msg, file=sys.stderr)
    sys.exit(1)


def find_conv(explicit):
    if explicit:
        if os.path.isfile(explicit):
            return explicit
        die("找不到 lv_font_conv: %s" % explicit)
    env = os.environ.get("LV_FONT_CONV")
    if env and os.path.isfile(env):
        return env
    which = shutil.which("lv_font_conv")
    if which:
        return which
    here = os.path.dirname(os.path.abspath(__file__))
    for cand in ("/tmp/opencode/node_modules/.bin/lv_font_conv",
                 os.path.join(here, "node_modules/.bin/lv_font_conv")):
        if os.path.isfile(cand):
            return cand
    die("未找到 lv_font_conv。请先安装 (npm install lv_font_conv), 或用 --gen 提供已生成的 C 文件")


def run_conv(conv, ttf, size, bpp, symbols):
    tmpdir = tempfile.mkdtemp(prefix="lvfont_")
    out = os.path.join(tmpdir, "gen.c")
    cmd = [conv, "--bpp", str(bpp), "--size", str(size), "--no-compress",
           "--font", ttf, "--symbols", symbols, "--format", "lvgl", "-o", out]
    print("运行: " + " ".join(cmd))
    r = subprocess.run(cmd, capture_output=True, text=True)
    if r.returncode != 0:
        die("lv_font_conv 执行失败:\n" + r.stdout + r.stderr)
    return out


def extract_array(text, decl):
    """定位数组 { 后内容起点, 返回 (内容起点, '\\n};' 的位置)"""
    m = re.search(re.escape(decl) + r'\s*\[[^\]]*\]\s*=\s*\{', text)
    if not m:
        die("在文件中找不到数组: " + decl)
    return m.end(), text.index("\n};", m.end())


def parse_gen_glyphs(text):
    """解析生成文件: 返回 [(code, char, bytes列表)] 按 unicode 升序"""
    start, end = extract_array(text, "glyph_bitmap")
    body = text[start:end]
    comments = list(GLYPH_COMMENT_RE.finditer(body))
    glyphs = []
    for i, m in enumerate(comments):
        code = int(m.group(1), 16)
        seg_start = m.end()
        seg_end = comments[i + 1].start() if i + 1 < len(comments) else len(body)
        tokens = TOKEN_RE.findall(body[seg_start:seg_end])
        glyphs.append((code, chr(code), [int(t, 16) for t in tokens]))
    glyphs.sort(key=lambda g: g[0])
    return glyphs


def parse_dsc(text):
    """解析 glyph_dsc 数组, 按数组顺序返回 dict 列表(含 id=0 保留项)"""
    start, end = extract_array(text, "glyph_dsc")
    body = text[start:end]
    return [dict(bitmap_index=int(m.group(1)), adv_w=int(m.group(2)),
                 box_w=int(m.group(3)), box_h=int(m.group(4)),
                 ofs_x=int(m.group(5)), ofs_y=int(m.group(6)))
            for m in DSC_RE.finditer(body)]


def parse_cmaps(text):
    """解析 cmaps, 返回 (稀疏cmap字典, 各字段在文本中的绝对位置)"""
    m = re.search(r'static const lv_font_fmt_txt_cmap_t cmaps\[\] =', text)
    if not m:
        die("文件中找不到 cmaps")
    end = text.index("\n};", m.end())
    seg = text[m.end():end]
    for fm in CMAP_FIRST_LINE_RE.finditer(seg):
        after = seg[fm.end():]
        nxt = re.search(r'\.range_start = \d+', after)
        tail = after if not nxt else after[:nxt.start()]
        if 'SPARSE' in tail:
            llen_m = LISTLEN_RE.search(tail)
            cmap = dict(range_start=int(fm.group(1)), range_length=int(fm.group(2)),
                        glyph_id_start=int(fm.group(3)), list_length=int(llen_m.group(1)))
            off = m.end()
            spans = {
                'range_start': (off + fm.start(1), off + fm.end(1)),
                'range_length': (off + fm.start(2), off + fm.end(2)),
                'glyph_id_start': (off + fm.start(3), off + fm.end(3)),
                'list_length': (off + fm.end() + llen_m.start(1), off + fm.end() + llen_m.end(1)),
            }
            return cmap, spans
    die("文件中没有 SPARSE 型 cmap")


def find_font_meta(text):
    bpp_m = re.search(r'\.bpp = (\d+)', text)
    lh_m = re.search(r'\.line_height = (\d+)', text)
    if not bpp_m or not lh_m:
        die("文件中找不到 .bpp 或 .line_height")
    return int(bpp_m.group(1)), int(lh_m.group(1))


def glyph_size(dsc, bpp):
    return math.ceil(dsc["box_w"] * dsc["box_h"] * bpp / 8)


def splice(target_path, glyphs, bpp):
    """glyphs: [(code, char, bytes列表, dsc)], 修改目标文件"""
    with open(target_path, "r", encoding="utf-8") as f:
        text = f.read()

    cmap, spans = parse_cmaps(text)
    rs, rl, gs = cmap["range_start"], cmap["range_length"], cmap["glyph_id_start"]

    bm_start, bm_end = extract_array(text, "glyph_bitmap")
    dsc_start, dsc_end = extract_array(text, "glyph_dsc")
    ul_m = re.search(r'unicode_list_1\[\] = \{', text)
    if not ul_m:
        die("目标文件中找不到 unicode_list_1")
    ul_end = text.index("\n};", ul_m.end())

    existing = [int(t, 16) for t in TOKEN_RE.findall(text[ul_m.end():ul_end])]
    dsc_body = text[dsc_start:dsc_end]
    dsc_entries = [(m.start(), m.end(), dict(bitmap_index=int(m.group(1)), adv_w=int(m.group(2)),
                                             box_w=int(m.group(3)), box_h=int(m.group(4)),
                                             ofs_x=int(m.group(5)), ofs_y=int(m.group(6))))
                   for m in DSC_RE.finditer(dsc_body)]
    if len(dsc_entries) != len(existing) + gs:
        die("解析异常: glyph_dsc 条目数(%d) 与 预期(%d) 不符" % (len(dsc_entries), len(existing) + gs))

    # ---- 自动扩展 cmap 范围 ----
    # 稀疏 cmap 的 unicode_list 偏移是 uint16_t(0-65535), 中文字符的 rcp 必然在此范围内,
    # 无需新增 cmap, 只需扩展 range_start(向下)/range_length(向上)
    new_rs = min(rs, min((code for code, ch, bts, dsc in glyphs), default=rs))
    delta = rs - new_rs
    if delta:
        existing = [v + delta for v in existing]
    rl_new = rl + delta

    # ---- 生成插入记录 ----
    records = []  # (k, rcp, code, char, bytes, dsc)
    for code, ch, bts, dsc in glyphs:
        rcp = code - new_rs
        if rcp < 0:
            die("字符 %s (U+%04X) 的 rcp=0x%x 为负, 无法放入 uint16 偏移" % (ch, code, rcp))
        if rcp > 65535:
            die("字符 %s (U+%04X) 的 rcp=0x%x 超过 uint16 上限, 需要新增独立 cmap(本脚本不支持)" % (ch, code, rcp))
        if rcp > rl_new:
            rl_new = rcp
        k = bisect.bisect_left(existing, rcp)
        if k < len(existing) and existing[k] == rcp:
            print("跳过: %s (U+%04X) 已存在于字体中" % (ch, code))
            continue
        expect = math.ceil(dsc["box_w"] * dsc["box_h"] * bpp / 8)
        if len(bts) != expect:
            die("字符 %s 位图字节数校验失败: 期望 %d, 实际 %d (生成参数与目标文件不一致?)" % (ch, expect, len(bts)))
        records.append((k, rcp, code, ch, bts, dsc))
    if not records:
        print("没有需要添加的字符")
        return []

    if new_rs != rs:
        print("cmap range_start 扩展: U+%04X → U+%04X" % (rs, new_rs))
    if rl_new > rl + delta:
        print("cmap range_length 扩展: %d → %d" % (rl, rl_new))
    rs = new_rs
    rl = rl_new

    # 多个新字可能落在同一区间: 按最终合并列表重新计算每个字的下标
    new_list = sorted(existing + [r[1] for r in records])
    pos_of = {v: i for i, v in enumerate(new_list)}
    records = [(pos_of[rcp], rcp, code, ch, bts, dsc) for k, rcp, code, ch, bts, dsc in records]
    records.sort(key=lambda r: r[0])

    # ---- 计算 glyph_id / bitmap_index ----
    total_bytes = len(TOKEN_RE.findall(text[bm_start:bm_end]))
    info = []
    for k, rcp, code, ch, bts, dsc in records:
        new_dsc = dict(dsc, bitmap_index=total_bytes)
        info.append((gs + k, rcp, code, ch, bts, new_dsc))
        total_bytes += len(bts)
    print("新增字符: " + ", ".join("%s → glyph_id #%d (bitmap 偏移 %d, %d 字节)" %
                                    (ch, gid, d["bitmap_index"], len(bts)) for gid, _, _, ch, bts, d in info))

    edits = []  # (pos, old_len, new_text), 按 pos 降序应用

    # ---- 1. glyph_bitmap: 末尾追加 ----
    ins = ""
    if text[bm_end - 1] != ',':
        ins += ","
    for idx, (gid, rcp, code, ch, bts, dsc) in enumerate(info):
        ins += "\n    /* U+%04X \"%s\" */\n" % (code, ch)
        lines = [bts[i:i + 8] for i in range(0, len(bts), 8)]
        last_global = (idx == len(info) - 1)
        for i, ln in enumerate(lines):
            comma = "," if (i < len(lines) - 1 or not last_global) else ""
            ins += "    " + ", ".join("0x%x" % b for b in ln) + comma + "\n"
    edits.append((bm_end, 0, ins))

    # ---- 2. glyph_dsc: 按 glyph_id 插入 ----
    # 锚点: 插入到"最终会落在下标 gid 的原条目"之前。
    # 原数组中被 gid 之前的插入挤占的条目, 其原下标 = gid - 之前插入数
    anchored = []  # (anchor_original_index, gid, dsc, ch)
    for i, (gid, rcp, code, ch, bts, dsc) in enumerate(info):
        anchored.append((gid - i, gid, dsc, ch))
    for anchor, gid, dsc, ch in sorted(anchored, key=lambda a: a[1], reverse=True):
        line = "    {.bitmap_index = %d, .adv_w = %d, .box_w = %d, .box_h = %d, .ofs_x = %d, .ofs_y = %d},\n" % (
            dsc["bitmap_index"], dsc["adv_w"], dsc["box_w"], dsc["box_h"], dsc["ofs_x"], dsc["ofs_y"])
        if anchor < len(dsc_entries):
            es, ee, _ = dsc_entries[anchor]
            line_start = dsc_body.rfind("\n", 0, es) + 1
            edits.append((dsc_start + line_start, 0, line))
        else:
            # 追加到数组末尾(每轮最多一个): 给上一行补逗号
            sep = "" if text[dsc_end - 1] == ',' else ","
            edits.append((dsc_end, 0, sep + "\n" + line))

    # ---- 3. unicode_list_1: 重写为升序 ----
    edits.append((ul_m.end() + 1, ul_end - ul_m.end() - 1, "    " + ", ".join("0x%x" % v for v in new_list)))

    # ---- 4. 稀疏 cmap 的 range_start / range_length / list_length ----
    edits.append((spans['range_start'][0], spans['range_start'][1] - spans['range_start'][0], str(rs)))
    edits.append((spans['range_length'][0], spans['range_length'][1] - spans['range_length'][0], str(rl)))
    edits.append((spans['list_length'][0], spans['list_length'][1] - spans['list_length'][0], str(len(new_list))))

    # ---- 5. 头注释 --symbols ----
    syms = "".join(chr(rs + v) for v in sorted(new_list))
    hdr_m = re.search(r'--symbols (\S+)', text)
    if hdr_m:
        edits.append((hdr_m.start(1), len(hdr_m.group(1)), syms))

    # ---- 应用编辑 ----
    for pos, old_len, new_text in sorted(edits, key=lambda e: e[0], reverse=True):
        text = text[:pos] + new_text + text[pos + old_len:]

    with open(target_path, "w", encoding="utf-8") as f:
        f.write(text)
    print("已写入: " + target_path)
    return info


def verify(target_path, added_chars, bpp, expected_dscs=None):
    with open(target_path, "r", encoding="utf-8") as f:
        text = f.read()
    errs = []

    cmap, _ = parse_cmaps(text)
    rs = cmap["range_start"]
    gs = cmap["glyph_id_start"]

    ul_m = re.search(r'unicode_list_1\[\] = \{', text)
    ul_end = text.index("\n};", ul_m.end())
    lst = [int(t, 16) for t in TOKEN_RE.findall(text[ul_m.end():ul_end])]
    if any(lst[i] >= lst[i + 1] for i in range(len(lst) - 1)):
        errs.append("unicode_list_1 不是严格升序")

    start, end = extract_array(text, "glyph_bitmap")
    total = len(TOKEN_RE.findall(text[start:end]))

    # C 语法粗查 1: 位图段去掉注释后, token 之间必须由逗号分隔
    body = re.sub(r'/\*.*?\*/', '', text[start:end])
    body = re.sub(r'\s+', '', body)
    if not re.fullmatch(r'0x[0-9a-fA-F]+(?:,0x[0-9a-fA-F]+)*', body):
        errs.append("glyph_bitmap 段 token 分隔有误(缺逗号?)")

    # C 语法粗查 2: glyph_dsc 每个条目必须是完整合法的单行结构
    dsc_start, dsc_end = extract_array(text, "glyph_dsc")
    dsc_sec = text[dsc_start:dsc_end]
    dsc_entry_re = re.compile(
        r'(?:\s*\{\.bitmap_index = \d+, \.adv_w = \d+, \.box_w = \d+, \.box_h = \d+, \.ofs_x = -?\d+, \.ofs_y = -?\d+\}'
        r'(?:\s*/\*.*\*/)?\s*,?\s*$|\s*$)')
    for i, line in enumerate(dsc_sec.splitlines(), 1):
        if not dsc_entry_re.match(line):
            errs.append("glyph_dsc 第 %d 行格式异常: %s" % (i, line.strip()[:60]))
    dscs = parse_dsc(text)
    ranges = []
    size_sum = 0
    for i, d in enumerate(dscs):
        if i == 0:
            continue  # 保留项
        size = math.ceil(d["box_w"] * d["box_h"] * bpp / 8)
        if d["bitmap_index"] < 0 or d["bitmap_index"] + size > total:
            errs.append("glyph_dsc[%d].bitmap_index=%d 越界 (总字节 %d)" % (i, d["bitmap_index"], total))
        ranges.append((d["bitmap_index"], d["bitmap_index"] + size))
        size_sum += size
    if size_sum != total:
        errs.append("glyph_bitmap 总字节数 %d, 字形字节和 %d, 不一致" % (total, size_sum))
    ranges.sort()
    for i in range(len(ranges) - 1):
        if ranges[i][1] > ranges[i + 1][0]:
            errs.append("字形位图区间重叠: %r 与 %r" % (ranges[i], ranges[i + 1]))

    for ch in added_chars:
        rcp = ord(ch) - rs
        if rcp not in lst:
            errs.append("字符 %s (rcp=0x%x) 未出现在 unicode_list_1" % (ch, rcp))
            continue
        gid = gs + lst.index(rcp)
        if gid >= len(dscs):
            errs.append("字符 %s 的 glyph_id %d 超出 dsc 数组" % (ch, gid))
            continue
        if expected_dscs and ch in expected_dscs:
            d = dscs[gid]
            e = expected_dscs[ch]
            for f in ("adv_w", "box_w", "box_h", "ofs_x", "ofs_y"):
                if d[f] != e[f]:
                    errs.append("字符 %s glyph_id %d 的 %s = %d, 期望 %d" % (ch, gid, f, d[f], e[f]))
                if f == "box_w" and d["box_w"] > 0 and d["box_h"] > 0 and d["bitmap_index"] + \
                        math.ceil(d["box_w"] * d["box_h"] * bpp / 8) > total:
                    errs.append("字符 %s glyph_id %d 位图越界" % (ch, gid))

    if errs:
        die("自校验失败:\n  " + "\n  ".join(errs))
    print("自校验通过: unicode_list 升序 / 位图区间无重叠越界 / 字节总数一致 / %d 个请求字符全部可查" % len(added_chars))


def main():
    ap = argparse.ArgumentParser(description="LVGL 字体增量加字脚本")
    ap.add_argument("chars", nargs="+", help="要添加的字符(可多个, 已存在的自动跳过)")
    ap.add_argument("--font", required=True, help="TTF 字体文件(必填, 相对/绝对路径均可)")
    ap.add_argument("--out", required=True, help="目标 C 文件(必填, 相对/绝对路径均可)")
    ap.add_argument("--conv", help="lv_font_conv 可执行文件路径")
    ap.add_argument("--gen", help="已生成的 C 文件(含新字形), 跳过 lv_font_conv")
    ap.add_argument("--size", type=int, default=None, help="字号(默认取目标文件 line_height)")
    ap.add_argument("--bpp", type=int, default=None, help="位深(默认取目标文件 .bpp)")
    args = ap.parse_args()

    out = os.path.abspath(args.out)
    ttf = os.path.abspath(args.font)
    if not os.path.isfile(out):
        die("目标文件不存在: " + out)
    if not args.gen and not os.path.isfile(ttf):
        die("字体文件不存在: " + ttf)

    with open(out, "r", encoding="utf-8") as f:
        text = f.read()
    bpp, line_height = find_font_meta(text)
    size = args.size or line_height
    bpp = args.bpp or bpp

    print("输入字体: %s (%dpx, bpp=%d)" % (ttf, size, bpp))
    print("输出文件: %s" % out)
    print("添加字符: %s" % " ".join(args.chars))

    symbols = "".join(dict.fromkeys(args.chars))
    if args.gen:
        gen_path = args.gen
        if not os.path.isfile(gen_path):
            die("生成文件不存在: " + gen_path)
    else:
        conv = find_conv(args.conv)
        gen_path = run_conv(conv, ttf, size, bpp, symbols)

    with open(gen_path, "r", encoding="utf-8") as f:
        gen_text = f.read()
    glyphs = parse_gen_glyphs(gen_text)
    dscs = parse_dsc(gen_text)
    if len(glyphs) != len(dscs) - 1:
        die("生成文件解析异常: 字形数 %d 与 dsc 数 %d 不匹配" % (len(glyphs), len(dscs) - 1))
    got = {ch for _, ch, _ in glyphs}
    missing = set(args.chars) - got
    if missing:
        die("以下字符在字体 %s 中不存在: %s" % (os.path.basename(ttf), "".join(sorted(missing))))

    glyphs = [(code, ch, bts, dscs[i + 1]) for i, (code, ch, bts) in enumerate(glyphs)]
    info = splice(out, glyphs, bpp)
    expected = {ch: dsc for gid, rcp, code, ch, bts, dsc in info}
    verify(out, list(dict.fromkeys(args.chars)), bpp, expected)
    print("完成: 新增 %d 个字符 → %s" % (len(info), out))


if __name__ == "__main__":
    main()