Files
2026-08-31 15:05:13 +08:00

150 lines
4.5 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
从 Unifont hex 数据中提取中文 16×16 点阵,生成 ESP32 可用的 C++ 数组。
用法:
python3 tools/gen_font.py <unifont.hex.gz>
输出:
main/font16cn_data.cpp
默认收录 GB2312 一级字(3,755 个常用字)+ 自定义扩展字符,
覆盖日常中文 99.7% 以上。如需添加额外字符,在 EXTRA_CHARS 中追加即可。
"""
import gzip
import sys
import os
# ========== GB2312 一级字 Unicode 码点列表 ==========
# GB2312 一级字包含 3,755 个常用字,按拼音排序
# 以下列表通过 GB2312 区位码 16-55 区转换得到
def _gb2312_level1_codepoints():
"""生成 GB2312 一级字(16-55 区)的 Unicode 码点集合"""
import codecs
codepoints = set()
for qu in range(16, 56): # 16~55 区
for wei in range(1, 95): # 1~94 位
b1 = qu + 0xA0
b2 = wei + 0xA0
try:
ch = bytes([b1, b2]).decode('gb2312')
codepoints.add(ord(ch))
except (UnicodeDecodeError, ValueError):
continue
return codepoints
# ========== 额外扩展字符(不在 GB2312 一级字中但项目需要的字符)==========
EXTRA_CHARS = "" # 在这里追加项目需要的特殊字符
def load_unifont_hex(hex_path):
"""加载 Unifont hex 文件,返回 {codepoint: bitmap_bytes} 字典"""
font_map = {}
opener = gzip.open if hex_path.endswith('.gz') else open
with opener(hex_path, 'rt', encoding='ascii') as f:
for line in f:
line = line.strip()
if not line or ':' not in line:
continue
cp_str, hex_data = line.split(':', 1)
cp = int(cp_str, 16)
font_map[cp] = bytes.fromhex(hex_data)
return font_map
def char_to_utf8_bytes(cp):
"""将 Unicode 码点转为 UTF-8 的 3 个字节(仅支持 U+0800~U+FFFF)"""
assert 0x0800 <= cp <= 0xFFFF
b1 = 0xE0 | (cp >> 12)
b2 = 0x80 | ((cp >> 6) & 0x3F)
b3 = 0x80 | (cp & 0x3F)
return (b1, b2, b3)
def generate_cpp(font_map, output_path):
"""生成 C++ 源文件"""
# 合并 GB2312 一级字 + 额外字符
target_cps = _gb2312_level1_codepoints()
for ch in EXTRA_CHARS:
target_cps.add(ord(ch))
entries = []
missing = 0
# 按 Unicode 排序(确保二分查找有效)
for cp in sorted(target_cps):
if cp not in font_map:
missing += 1
continue
bitmap = font_map[cp]
# Unifont 16×16 汉字 = 32 字节
if len(bitmap) != 32:
missing += 1
continue
utf8 = char_to_utf8_bytes(cp)
hex_str = ', '.join(f'0x{b:02X}' for b in bitmap)
entries.append(
f' // U+{cp:04X} "{chr(cp)}"\n'
f' {{ {{ 0x{utf8[0]:02X}, 0x{utf8[1]:02X}, 0x{utf8[2]:02X} }},\n'
f' {{ {hex_str} }} }}'
)
total_bytes = len(entries) * 35 # 每个条目约 35 字节
cpp_content = f'''\
/**
* @file font16cn_data.cpp
* @brief Unifont 16×16 中文点阵数据(自动生成,请勿手动编辑)。
*
* 由 tools/gen_font.py 从 Unifont hex 数据自动提取。
* 收录 {len(entries)} 个常用中文字符(GB2312 一级字)。
* 预计占用 Flash: ~{total_bytes // 1024} KB
*
* 如需添加字符,编辑 gen_font.py 中的 EXTRA_CHARS 并重新运行。
*/
#include "ChineseFont.h"
static const UTF8_Char kFont16CN_Table[] = {{
{(","+chr(10)).join(entries)}
}};
const UTF8_Font Font16CN = {{
kFont16CN_Table,
{len(entries)}, // 字符数
8, // ASCII 字符宽度(半角)
16, // 中文字符宽度
16 // 字符高度
}};
'''
os.makedirs(os.path.dirname(output_path), exist_ok=True)
with open(output_path, 'w', encoding='utf-8') as f:
f.write(cpp_content)
print(f"✅ 已生成 {output_path}")
print(f" 收录: {len(entries)} 个汉字")
print(f" 未找到: {missing} 个")
print(f" 预计 Flash: ~{total_bytes // 1024} KB")
def main():
if len(sys.argv) < 2:
print(f"用法: {sys.argv[0]} <unifont.hex 或 unifont.hex.gz>")
sys.exit(1)
hex_path = sys.argv[1]
script_dir = os.path.dirname(os.path.abspath(__file__))
output_path = os.path.join(script_dir, '..', 'components', 'chinese_font', 'src', 'font16cn_data.cpp')
print(f"📖 加载 Unifont 数据: {hex_path}")
font_map = load_unifont_hex(hex_path)
print(f" 共加载 {len(font_map)} 个码点")
generate_cpp(font_map, output_path)
if __name__ == '__main__':
main()