python脚本
#!/usr/bin/env python3
"""向 lv_font_conv 生成的 LVGL 字体 C 文件增量添加字符。
用法(--font 输入字体、--out 输出文件为必填, 相对/绝对路径均可):
python3 add_font_chars.py --font simhei.ttf --out lvgl/src/font/songti_20.c 新
python3 add_font_chars.py --font /abs/path/a.ttf --out /abs/path/b.c 新 操 满 # 一次添加多个字
python3 add_font_chars.py --font simhei.ttf --out out.c --gen generated.c 新 # 用官网生成的文件代替 lv_font_conv
--gen 是数据源替换选项:跳过本地 lv_font_conv,改用你自己准备的生成文件来提供新字的位图数据。
说明: 已存在于字体中的字符会自动跳过并打印日志, 不会报错。
超出当前 cmap 范围的字符会自动扩展 range_start/range_length(中文字符的 rcp 必然在
uint16 范围内, 无需新增 cmap)。
原理:
1. 用 lv_font_conv 只生成新增字符的字形数据(与目标文件同 bpp/size/无压缩, 格式保证一致)
2. 解析出每个新字的位图字节和字形参数(adv_w/box_w/box_h/ofs_x/ofs_y)
3. 按 LVGL fmt_txt 规则拼入目标文件:
- glyph_bitmap: 字节追加到数组末尾, bitmap_index = 当前总字节数
- glyph_dsc: 按 glyph_id (= glyph_id_start + 在 unicode_list 中的下标)插入, 后续条目顺延
- unicode_list: 保持升序插入(LVGL 二分查找要求), list_length 同步加
- cmap: range_start/range_length 按需自动扩展
4. 自校验: 升序性 / 位图区间无重叠越界 / 字节数一致 / 字形映射正确
"""
import argparse
import bisect
import math
import os
import re
import shutil
import subprocess
import sys
import tempfile
TOKEN_RE = re.compile(r'0x[0-9a-fA-F]+')
DSC_RE = re.compile(
r'\.bitmap_index = (\d+),\s*\.adv_w = (\d+),\s*\.box_w = (\d+),\s*\.box_h = (\d+),\s*\.ofs_x = (-?\d+),\s*\.ofs_y = (-?\d+)')
GLYPH_COMMENT_RE = re.compile(r'/\* U\+([0-9A-Fa-f]{1,6}) "([^"]*)" \*/')
CMAP_FIRST_LINE_RE = re.compile(
r'\.range_start = (\d+),\s*\.range_length = (\d+),\s*\.glyph_id_start = (\d+),')
LISTLEN_RE = re.compile(r'\.list_length = (\d+)')
def die(msg):
print("错误: " + msg, file=sys.stderr)
sys.exit(1)
def find_conv(explicit):
if explicit:
if os.path.isfile(explicit):
return explicit
die("找不到 lv_font_conv: %s" % explicit)
env = os.environ.get("LV_FONT_CONV")
if env and os.path.isfile(env):
return env
which = shutil.which("lv_font_conv")
if which:
return which
here = os.path.dirname(os.path.abspath(__file__))
for cand in ("/tmp/opencode/node_modules/.bin/lv_font_conv",
os.path.join(here, "node_modules/.bin/lv_font_conv")):
if os.path.isfile(cand):
return cand
die("未找到 lv_font_conv。请先安装 (npm install lv_font_conv), 或用 --gen 提供已生成的 C 文件")
def run_conv(conv, ttf, size, bpp, symbols):
tmpdir = tempfile.mkdtemp(prefix="lvfont_")
out = os.path.join(tmpdir, "gen.c")
cmd = [conv, "--bpp", str(bpp), "--size", str(size), "--no-compress",
"--font", ttf, "--symbols", symbols, "--format", "lvgl", "-o", out]
print("运行: " + " ".join(cmd))
r = subprocess.run(cmd, capture_output=True, text=True)
if r.returncode != 0:
die("lv_font_conv 执行失败:\n" + r.stdout + r.stderr)
return out
def extract_array(text, decl):
"""定位数组 { 后内容起点, 返回 (内容起点, '\\n};' 的位置)"""
m = re.search(re.escape(decl) + r'\s*\[[^\]]*\]\s*=\s*\{', text)
if not m:
die("在文件中找不到数组: " + decl)
return m.end(), text.index("\n};", m.end())
def parse_gen_glyphs(text):
"""解析生成文件: 返回 [(code, char, bytes列表)] 按 unicode 升序"""
start, end = extract_array(text, "glyph_bitmap")
body = text[start:end]
comments = list(GLYPH_COMMENT_RE.finditer(body))
glyphs = []
for i, m in enumerate(comments):
code = int(m.group(1), 16)
seg_start = m.end()
seg_end = comments[i + 1].start() if i + 1 < len(comments) else len(body)
tokens = TOKEN_RE.findall(body[seg_start:seg_end])
glyphs.append((code, chr(code), [int(t, 16) for t in tokens]))
glyphs.sort(key=lambda g: g[0])
return glyphs
def parse_dsc(text):
"""解析 glyph_dsc 数组, 按数组顺序返回 dict 列表(含 id=0 保留项)"""
start, end = extract_array(text, "glyph_dsc")
body = text[start:end]
return [dict(bitmap_index=int(m.group(1)), adv_w=int(m.group(2)),
box_w=int(m.group(3)), box_h=int(m.group(4)),
ofs_x=int(m.group(5)), ofs_y=int(m.group(6)))
for m in DSC_RE.finditer(body)]
def parse_cmaps(text):
"""解析 cmaps, 返回 (稀疏cmap字典, 各字段在文本中的绝对位置)"""
m = re.search(r'static const lv_font_fmt_txt_cmap_t cmaps\[\] =', text)
if not m:
die("文件中找不到 cmaps")
end = text.index("\n};", m.end())
seg = text[m.end():end]
for fm in CMAP_FIRST_LINE_RE.finditer(seg):
after = seg[fm.end():]
nxt = re.search(r'\.range_start = \d+', after)
tail = after if not nxt else after[:nxt.start()]
if 'SPARSE' in tail:
llen_m = LISTLEN_RE.search(tail)
cmap = dict(range_start=int(fm.group(1)), range_length=int(fm.group(2)),
glyph_id_start=int(fm.group(3)), list_length=int(llen_m.group(1)))
off = m.end()
spans = {
'range_start': (off + fm.start(1), off + fm.end(1)),
'range_length': (off + fm.start(2), off + fm.end(2)),
'glyph_id_start': (off + fm.start(3), off + fm.end(3)),
'list_length': (off + fm.end() + llen_m.start(1), off + fm.end() + llen_m.end(1)),
}
return cmap, spans
die("文件中没有 SPARSE 型 cmap")
def find_font_meta(text):
bpp_m = re.search(r'\.bpp = (\d+)', text)
lh_m = re.search(r'\.line_height = (\d+)', text)
if not bpp_m or not lh_m:
die("文件中找不到 .bpp 或 .line_height")
return int(bpp_m.group(1)), int(lh_m.group(1))
def glyph_size(dsc, bpp):
return math.ceil(dsc["box_w"] * dsc["box_h"] * bpp / 8)
def splice(target_path, glyphs, bpp):
"""glyphs: [(code, char, bytes列表, dsc)], 修改目标文件"""
with open(target_path, "r", encoding="utf-8") as f:
text = f.read()
cmap, spans = parse_cmaps(text)
rs, rl, gs = cmap["range_start"], cmap["range_length"], cmap["glyph_id_start"]
bm_start, bm_end = extract_array(text, "glyph_bitmap")
dsc_start, dsc_end = extract_array(text, "glyph_dsc")
ul_m = re.search(r'unicode_list_1\[\] = \{', text)
if not ul_m:
die("目标文件中找不到 unicode_list_1")
ul_end = text.index("\n};", ul_m.end())
existing = [int(t, 16) for t in TOKEN_RE.findall(text[ul_m.end():ul_end])]
dsc_body = text[dsc_start:dsc_end]
dsc_entries = [(m.start(), m.end(), dict(bitmap_index=int(m.group(1)), adv_w=int(m.group(2)),
box_w=int(m.group(3)), box_h=int(m.group(4)),
ofs_x=int(m.group(5)), ofs_y=int(m.group(6))))
for m in DSC_RE.finditer(dsc_body)]
if len(dsc_entries) != len(existing) + gs:
die("解析异常: glyph_dsc 条目数(%d) 与 预期(%d) 不符" % (len(dsc_entries), len(existing) + gs))
# ---- 自动扩展 cmap 范围 ----
# 稀疏 cmap 的 unicode_list 偏移是 uint16_t(0-65535), 中文字符的 rcp 必然在此范围内,
# 无需新增 cmap, 只需扩展 range_start(向下)/range_length(向上)
new_rs = min(rs, min((code for code, ch, bts, dsc in glyphs), default=rs))
delta = rs - new_rs
if delta:
existing = [v + delta for v in existing]
rl_new = rl + delta
# ---- 生成插入记录 ----
records = [] # (k, rcp, code, char, bytes, dsc)
for code, ch, bts, dsc in glyphs:
rcp = code - new_rs
if rcp < 0:
die("字符 %s (U+%04X) 的 rcp=0x%x 为负, 无法放入 uint16 偏移" % (ch, code, rcp))
if rcp > 65535:
die("字符 %s (U+%04X) 的 rcp=0x%x 超过 uint16 上限, 需要新增独立 cmap(本脚本不支持)" % (ch, code, rcp))
if rcp > rl_new:
rl_new = rcp
k = bisect.bisect_left(existing, rcp)
if k < len(existing) and existing[k] == rcp:
print("跳过: %s (U+%04X) 已存在于字体中" % (ch, code))
continue
expect = math.ceil(dsc["box_w"] * dsc["box_h"] * bpp / 8)
if len(bts) != expect:
die("字符 %s 位图字节数校验失败: 期望 %d, 实际 %d (生成参数与目标文件不一致?)" % (ch, expect, len(bts)))
records.append((k, rcp, code, ch, bts, dsc))
if not records:
print("没有需要添加的字符")
return []
if new_rs != rs:
print("cmap range_start 扩展: U+%04X → U+%04X" % (rs, new_rs))
if rl_new > rl + delta:
print("cmap range_length 扩展: %d → %d" % (rl, rl_new))
rs = new_rs
rl = rl_new
# 多个新字可能落在同一区间: 按最终合并列表重新计算每个字的下标
new_list = sorted(existing + [r[1] for r in records])
pos_of = {v: i for i, v in enumerate(new_list)}
records = [(pos_of[rcp], rcp, code, ch, bts, dsc) for k, rcp, code, ch, bts, dsc in records]
records.sort(key=lambda r: r[0])
# ---- 计算 glyph_id / bitmap_index ----
total_bytes = len(TOKEN_RE.findall(text[bm_start:bm_end]))
info = []
for k, rcp, code, ch, bts, dsc in records:
new_dsc = dict(dsc, bitmap_index=total_bytes)
info.append((gs + k, rcp, code, ch, bts, new_dsc))
total_bytes += len(bts)
print("新增字符: " + ", ".join("%s → glyph_id #%d (bitmap 偏移 %d, %d 字节)" %
(ch, gid, d["bitmap_index"], len(bts)) for gid, _, _, ch, bts, d in info))
edits = [] # (pos, old_len, new_text), 按 pos 降序应用
# ---- 1. glyph_bitmap: 末尾追加 ----
ins = ""
if text[bm_end - 1] != ',':
ins += ","
for idx, (gid, rcp, code, ch, bts, dsc) in enumerate(info):
ins += "\n /* U+%04X \"%s\" */\n" % (code, ch)
lines = [bts[i:i + 8] for i in range(0, len(bts), 8)]
last_global = (idx == len(info) - 1)
for i, ln in enumerate(lines):
comma = "," if (i < len(lines) - 1 or not last_global) else ""
ins += " " + ", ".join("0x%x" % b for b in ln) + comma + "\n"
edits.append((bm_end, 0, ins))
# ---- 2. glyph_dsc: 按 glyph_id 插入 ----
# 锚点: 插入到"最终会落在下标 gid 的原条目"之前。
# 原数组中被 gid 之前的插入挤占的条目, 其原下标 = gid - 之前插入数
anchored = [] # (anchor_original_index, gid, dsc, ch)
for i, (gid, rcp, code, ch, bts, dsc) in enumerate(info):
anchored.append((gid - i, gid, dsc, ch))
for anchor, gid, dsc, ch in sorted(anchored, key=lambda a: a[1], reverse=True):
line = " {.bitmap_index = %d, .adv_w = %d, .box_w = %d, .box_h = %d, .ofs_x = %d, .ofs_y = %d},\n" % (
dsc["bitmap_index"], dsc["adv_w"], dsc["box_w"], dsc["box_h"], dsc["ofs_x"], dsc["ofs_y"])
if anchor < len(dsc_entries):
es, ee, _ = dsc_entries[anchor]
line_start = dsc_body.rfind("\n", 0, es) + 1
edits.append((dsc_start + line_start, 0, line))
else:
# 追加到数组末尾(每轮最多一个): 给上一行补逗号
sep = "" if text[dsc_end - 1] == ',' else ","
edits.append((dsc_end, 0, sep + "\n" + line))
# ---- 3. unicode_list_1: 重写为升序 ----
edits.append((ul_m.end() + 1, ul_end - ul_m.end() - 1, " " + ", ".join("0x%x" % v for v in new_list)))
# ---- 4. 稀疏 cmap 的 range_start / range_length / list_length ----
edits.append((spans['range_start'][0], spans['range_start'][1] - spans['range_start'][0], str(rs)))
edits.append((spans['range_length'][0], spans['range_length'][1] - spans['range_length'][0], str(rl)))
edits.append((spans['list_length'][0], spans['list_length'][1] - spans['list_length'][0], str(len(new_list))))
# ---- 5. 头注释 --symbols ----
syms = "".join(chr(rs + v) for v in sorted(new_list))
hdr_m = re.search(r'--symbols (\S+)', text)
if hdr_m:
edits.append((hdr_m.start(1), len(hdr_m.group(1)), syms))
# ---- 应用编辑 ----
for pos, old_len, new_text in sorted(edits, key=lambda e: e[0], reverse=True):
text = text[:pos] + new_text + text[pos + old_len:]
with open(target_path, "w", encoding="utf-8") as f:
f.write(text)
print("已写入: " + target_path)
return info
def verify(target_path, added_chars, bpp, expected_dscs=None):
with open(target_path, "r", encoding="utf-8") as f:
text = f.read()
errs = []
cmap, _ = parse_cmaps(text)
rs = cmap["range_start"]
gs = cmap["glyph_id_start"]
ul_m = re.search(r'unicode_list_1\[\] = \{', text)
ul_end = text.index("\n};", ul_m.end())
lst = [int(t, 16) for t in TOKEN_RE.findall(text[ul_m.end():ul_end])]
if any(lst[i] >= lst[i + 1] for i in range(len(lst) - 1)):
errs.append("unicode_list_1 不是严格升序")
start, end = extract_array(text, "glyph_bitmap")
total = len(TOKEN_RE.findall(text[start:end]))
# C 语法粗查 1: 位图段去掉注释后, token 之间必须由逗号分隔
body = re.sub(r'/\*.*?\*/', '', text[start:end])
body = re.sub(r'\s+', '', body)
if not re.fullmatch(r'0x[0-9a-fA-F]+(?:,0x[0-9a-fA-F]+)*', body):
errs.append("glyph_bitmap 段 token 分隔有误(缺逗号?)")
# C 语法粗查 2: glyph_dsc 每个条目必须是完整合法的单行结构
dsc_start, dsc_end = extract_array(text, "glyph_dsc")
dsc_sec = text[dsc_start:dsc_end]
dsc_entry_re = re.compile(
r'(?:\s*\{\.bitmap_index = \d+, \.adv_w = \d+, \.box_w = \d+, \.box_h = \d+, \.ofs_x = -?\d+, \.ofs_y = -?\d+\}'
r'(?:\s*/\*.*\*/)?\s*,?\s*$|\s*$)')
for i, line in enumerate(dsc_sec.splitlines(), 1):
if not dsc_entry_re.match(line):
errs.append("glyph_dsc 第 %d 行格式异常: %s" % (i, line.strip()[:60]))
dscs = parse_dsc(text)
ranges = []
size_sum = 0
for i, d in enumerate(dscs):
if i == 0:
continue # 保留项
size = math.ceil(d["box_w"] * d["box_h"] * bpp / 8)
if d["bitmap_index"] < 0 or d["bitmap_index"] + size > total:
errs.append("glyph_dsc[%d].bitmap_index=%d 越界 (总字节 %d)" % (i, d["bitmap_index"], total))
ranges.append((d["bitmap_index"], d["bitmap_index"] + size))
size_sum += size
if size_sum != total:
errs.append("glyph_bitmap 总字节数 %d, 字形字节和 %d, 不一致" % (total, size_sum))
ranges.sort()
for i in range(len(ranges) - 1):
if ranges[i][1] > ranges[i + 1][0]:
errs.append("字形位图区间重叠: %r 与 %r" % (ranges[i], ranges[i + 1]))
for ch in added_chars:
rcp = ord(ch) - rs
if rcp not in lst:
errs.append("字符 %s (rcp=0x%x) 未出现在 unicode_list_1" % (ch, rcp))
continue
gid = gs + lst.index(rcp)
if gid >= len(dscs):
errs.append("字符 %s 的 glyph_id %d 超出 dsc 数组" % (ch, gid))
continue
if expected_dscs and ch in expected_dscs:
d = dscs[gid]
e = expected_dscs[ch]
for f in ("adv_w", "box_w", "box_h", "ofs_x", "ofs_y"):
if d[f] != e[f]:
errs.append("字符 %s glyph_id %d 的 %s = %d, 期望 %d" % (ch, gid, f, d[f], e[f]))
if f == "box_w" and d["box_w"] > 0 and d["box_h"] > 0 and d["bitmap_index"] + \
math.ceil(d["box_w"] * d["box_h"] * bpp / 8) > total:
errs.append("字符 %s glyph_id %d 位图越界" % (ch, gid))
if errs:
die("自校验失败:\n " + "\n ".join(errs))
print("自校验通过: unicode_list 升序 / 位图区间无重叠越界 / 字节总数一致 / %d 个请求字符全部可查" % len(added_chars))
def main():
ap = argparse.ArgumentParser(description="LVGL 字体增量加字脚本")
ap.add_argument("chars", nargs="+", help="要添加的字符(可多个, 已存在的自动跳过)")
ap.add_argument("--font", required=True, help="TTF 字体文件(必填, 相对/绝对路径均可)")
ap.add_argument("--out", required=True, help="目标 C 文件(必填, 相对/绝对路径均可)")
ap.add_argument("--conv", help="lv_font_conv 可执行文件路径")
ap.add_argument("--gen", help="已生成的 C 文件(含新字形), 跳过 lv_font_conv")
ap.add_argument("--size", type=int, default=None, help="字号(默认取目标文件 line_height)")
ap.add_argument("--bpp", type=int, default=None, help="位深(默认取目标文件 .bpp)")
args = ap.parse_args()
out = os.path.abspath(args.out)
ttf = os.path.abspath(args.font)
if not os.path.isfile(out):
die("目标文件不存在: " + out)
if not args.gen and not os.path.isfile(ttf):
die("字体文件不存在: " + ttf)
with open(out, "r", encoding="utf-8") as f:
text = f.read()
bpp, line_height = find_font_meta(text)
size = args.size or line_height
bpp = args.bpp or bpp
print("输入字体: %s (%dpx, bpp=%d)" % (ttf, size, bpp))
print("输出文件: %s" % out)
print("添加字符: %s" % " ".join(args.chars))
symbols = "".join(dict.fromkeys(args.chars))
if args.gen:
gen_path = args.gen
if not os.path.isfile(gen_path):
die("生成文件不存在: " + gen_path)
else:
conv = find_conv(args.conv)
gen_path = run_conv(conv, ttf, size, bpp, symbols)
with open(gen_path, "r", encoding="utf-8") as f:
gen_text = f.read()
glyphs = parse_gen_glyphs(gen_text)
dscs = parse_dsc(gen_text)
if len(glyphs) != len(dscs) - 1:
die("生成文件解析异常: 字形数 %d 与 dsc 数 %d 不匹配" % (len(glyphs), len(dscs) - 1))
got = {ch for _, ch, _ in glyphs}
missing = set(args.chars) - got
if missing:
die("以下字符在字体 %s 中不存在: %s" % (os.path.basename(ttf), "".join(sorted(missing))))
glyphs = [(code, ch, bts, dscs[i + 1]) for i, (code, ch, bts) in enumerate(glyphs)]
info = splice(out, glyphs, bpp)
expected = {ch: dsc for gid, rcp, code, ch, bts, dsc in info}
verify(out, list(dict.fromkeys(args.chars)), bpp, expected)
print("完成: 新增 %d 个字符 → %s" % (len(info), out))
if __name__ == "__main__":
main()