Files
blog/scripts/subset-font-safe.py
T
Vaica 80940b0316 feat: 添加古风官职特殊字符 + 优化字体子集化脚本
字符添加:
- 添加破折号(—)
- 添加庶、尉、谏、簿、丞
- 确保古风官职文本完整显示

脚本优化:
- 修改subset-font-safe.py,保留现有字符
- 合并新提取的字符,避免覆盖手动添加的字符
- 字体大小:740.7 KB(减少39.6%)
2026-06-03 14:53:56 +08:00

228 lines
7.7 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
字体子集化脚本 - 安全版本
从HTML和CSS文件中提取实际使用的字符,生成优化的子集字体
"""
import os
import re
import sys
from fontTools.ttLib import TTFont
from fontTools.subset import Subsetter, Options
def extract_chars_from_files(directories, extensions=('.html', '.md', '.css')):
"""从文件中提取使用的字符"""
chars = set()
for directory in directories:
if not os.path.exists(directory):
print(f"⚠️ 目录不存在,跳过: {directory}")
continue
print(f"🔍 扫描目录: {directory}")
for root, dirs, files in os.walk(directory):
for file in files:
if any(file.endswith(ext) for ext in extensions):
filepath = os.path.join(root, file)
try:
with open(filepath, 'r', encoding='utf-8') as f:
content = f.read()
# 提取中文字符
chinese_chars = re.findall(r'[一-鿿]', content)
chars.update(chinese_chars)
# 提取中文标点
cjk_punct = re.findall(r'[ -〿＀-￯]', content)
chars.update(cjk_punct)
# 提取英文和数字
ascii_chars = re.findall(r'[a-zA-Z0-9]', content)
chars.update(ascii_chars)
# 提取常用英文标点
en_punct = set('!@#$%^&*()_+-=[]{}|;:,.<>?/\\`~"\'-')
chars.update(en_punct)
except Exception as e:
print(f"⚠️ 读取文件失败 {filepath}: {e}")
return chars
def subset_font(input_font_path, output_font_path, chars):
"""生成子集字体"""
print(f"\n📦 加载字体: {input_font_path}")
# 检查输入文件
if not os.path.exists(input_font_path):
print(f"❌ 字体文件不存在: {input_font_path}")
return False
# 加载原始字体
try:
font = TTFont(input_font_path)
except Exception as e:
print(f"❌ 加载字体失败: {e}")
return False
# 获取原始字符数
cmap = font.getBestCmap()
original_count = len(cmap) if cmap else 0
original_size = os.path.getsize(input_font_path)
print(f"📊 原始字体信息:")
print(f" 字符数: {original_count}")
print(f" 文件大小: {original_size / 1024:.1f} KB")
# 配置子集化选项
options = Options()
if output_font_path.endswith('.woff2'):
options.flavor = 'woff2'
elif output_font_path.endswith('.woff'):
options.flavor = 'woff'
options.desubroutinize = True
options.layout_features = ['kern', 'liga'] # 保留基本的排版特性
# 创建子集化器
subsetter = Subsetter(options=options)
# 填充字符集
char_text = ''.join(chars)
print(f"\n✂️ 正在生成子集字体...")
print(f" 提取的字符数: {len(chars)}")
try:
subsetter.populate(text=char_text)
subsetter.subset(font)
except Exception as e:
print(f"❌ 子集化失败: {e}")
return False
# 保存子集字体
try:
font.save(output_font_path)
except Exception as e:
print(f"❌ 保存字体失败: {e}")
return False
# 获取子集信息
subset_size = os.path.getsize(output_font_path)
subset_font_obj = TTFont(output_font_path)
subset_cmap = subset_font_obj.getBestCmap()
subset_count = len(subset_cmap) if subset_cmap else 0
# 计算优化效果
reduction = original_size - subset_size
percentage = (reduction / original_size) * 100 if original_size > 0 else 0
print(f"\n✅ 子集化完成!")
print(f"📊 优化结果:")
print(f" 子集字符数: {subset_count}")
print(f" 子集文件大小: {subset_size / 1024:.1f} KB")
print(f" 减少: {reduction / 1024:.1f} KB ({percentage:.1f}%)")
# 验证是否真的优化了
if subset_size >= original_size:
print(f"\n⚠️ 警告: 子集字体反而变大了!")
print(f" 原因可能是提取了太多字符")
print(f" 建议检查扫描范围或使用更小的字符集")
return False
return True
def main():
print("🔤 字体子集化工具(安全版本)")
print("=" * 50)
# 配置路径
font_dir = "themes/Ying/static/font"
input_font = os.path.join(font_dir, "zql-v2.woff2")
chars_file = os.path.join(font_dir, "used_chars.txt")
# 检查输入文件
if not os.path.exists(input_font):
print(f"❌ 字体文件不存在: {input_font}")
print(f" 请确保在博客根目录运行此脚本")
sys.exit(1)
# 读取现有的字符列表(如果存在)
existing_chars = set()
if os.path.exists(chars_file):
with open(chars_file, 'r', encoding='utf-8') as f:
existing_chars = set(f.read())
print(f"📖 读取现有字符: {len(existing_chars)} 个")
else:
print(f"⚠️ 现有字符文件不存在,将创建新文件")
# 扫描目录
scan_dirs = ["content", "layouts"]
# 检查public目录是否存在
if os.path.exists("public"):
scan_dirs.append("public")
print(f"✅ 找到public目录,将扫描构建后的HTML")
else:
print(f"⚠️ 未找到public目录,请先运行: hugo --destination=public")
print(f" 将只扫描content和layouts目录")
print(f"\n🔍 扫描目录: {', '.join(scan_dirs)}")
# 提取字符
new_chars = extract_chars_from_files(scan_dirs)
print(f"\n📝 从文件中提取了 {len(new_chars)} 个唯一字符")
# 合并字符(保留现有字符 + 新提取的字符)
all_chars = existing_chars | new_chars
print(f"📊 合并后字符数: {len(all_chars)}")
if len(all_chars) == 0:
print("❌ 未找到任何字符,请检查扫描目录")
sys.exit(1)
# 保存合并后的字符列表
with open(chars_file, 'w', encoding='utf-8') as f:
sorted_chars = sorted(all_chars)
f.write(''.join(sorted_chars))
print(f"💾 字符列表已保存到: {chars_file}")
# 显示部分字符(用于验证)
sample_chars = sorted(list(all_chars))[:50]
print(f"📋 前50个字符: {''.join(sample_chars)}")
# 生成woff2子集字体
output_woff2 = os.path.join(font_dir, "zql-v2-subset.woff2")
print(f"\n{'='*50}")
print(f"🎯 生成 woff2 子集字体...")
success_woff2 = subset_font(input_font, output_woff2, all_chars)
# 生成woff子集字体
output_woff = os.path.join(font_dir, "zql-v2-subset.woff")
print(f"\n{'='*50}")
print(f"🎯 生成 woff 子集字体...")
success_woff = subset_font(input_font, output_woff, all_chars)
# 总结
print(f"\n{'='*50}")
if success_woff2 and success_woff:
print(f"🎉 所有子集字体生成成功!")
# 显示最终文件大小
print(f"\n📂 最终文件:")
for f in [output_woff2, output_woff]:
if os.path.exists(f):
size = os.path.getsize(f)
print(f" {os.path.basename(f)}: {size / 1024:.1f} KB")
print(f"\n✅ 下一步:")
print(f" 1. 更新CSS字体声明(将zql-v2.woff2改为zql-v2-subset.woff2)")
print(f" 2. 重新构建Hugo: hugo server -D")
print(f" 3. 测试字体显示是否正常")
else:
print(f"⚠️ 部分字体生成失败或未优化")
print(f" 建议使用fallback方案(保留原始字体)")
if __name__ == "__main__":
main()