#!/usr/bin/env python3 """ 字体子集化脚本 - 安全版本 从HTML和CSS文件中提取实际使用的字符,生成优化的子集字体 """ import os import re import sys from fontTools.ttLib import TTFont from fontTools.subset import Subsetter, Options def extract_chars_from_files(directories, extensions=('.html', '.md', '.css')): """从文件中提取使用的字符""" chars = set() for directory in directories: if not os.path.exists(directory): print(f"⚠️ 目录不存在,跳过: {directory}") continue print(f"🔍 扫描目录: {directory}") for root, dirs, files in os.walk(directory): for file in files: if any(file.endswith(ext) for ext in extensions): filepath = os.path.join(root, file) try: with open(filepath, 'r', encoding='utf-8') as f: content = f.read() # 提取中文字符 chinese_chars = re.findall(r'[一-鿿]', content) chars.update(chinese_chars) # 提取中文标点 cjk_punct = re.findall(r'[ -〿＀-￯]', content) chars.update(cjk_punct) # 提取英文和数字 ascii_chars = re.findall(r'[a-zA-Z0-9]', content) chars.update(ascii_chars) # 提取常用英文标点 en_punct = set('!@#$%^&*()_+-=[]{}|;:,.<>?/\\`~"\'-') chars.update(en_punct) except Exception as e: print(f"⚠️ 读取文件失败 {filepath}: {e}") return chars def extract_chars_from_json(json_files): """从JSON数据文件中提取字符(用于友链、朋友圈等动态内容)""" chars = set() import json for json_file in json_files: if not os.path.exists(json_file): print(f"⚠️ JSON文件不存在,跳过: {json_file}") continue print(f"🔍 扫描JSON文件: {json_file}") try: with open(json_file, 'r', encoding='utf-8') as f: data = json.load(f) # 将整个JSON转换为字符串,提取所有字符 content = json.dumps(data, ensure_ascii=False) # 提取中文字符 chinese_chars = re.findall(r'[一-鿿]', content) chars.update(chinese_chars) # 提取中文标点 cjk_punct = re.findall(r'[ -〿＀-￯]', content) chars.update(cjk_punct) # 提取英文和数字 ascii_chars = re.findall(r'[a-zA-Z0-9]', content) chars.update(ascii_chars) print(f" ✅ 从 {json_file} 提取了 {len(chinese_chars)} 个中文字符") except Exception as e: print(f"⚠️ 读取JSON文件失败 {json_file}: {e}") return chars def fetch_and_extract_chars_from_api(urls): """从API接口获取数据并提取字符(用于动态内容如友链、朋友圈)""" chars = set() try: import urllib.request import json import ssl # 创建不验证SSL的上下文(用于内网或自签名证书) context = ssl._create_unverified_context() for url in urls: print(f"🌐 尝试从API获取: {url}") try: req = urllib.request.Request( url, headers={ 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36' } ) with urllib.request.urlopen(req, context=context, timeout=15) as response: data = response.read().decode('utf-8') try: json_data = json.loads(data) content = json.dumps(json_data, ensure_ascii=False) # 提取中文字符 chinese_chars = re.findall(r'[一-鿿]', content) chars.update(chinese_chars) print(f" ✅ 从API提取了 {len(chinese_chars)} 个中文字符") except json.JSONDecodeError: print(f" ⚠️ 响应不是有效的JSON,跳过") except urllib.error.URLError as e: print(f" ❌ API请求失败: {e}") except Exception as e: print(f" ❌ 处理API响应失败: {e}") except ImportError: print("⚠️ urllib不可用,跳过API扫描") return chars def subset_font(input_font_path, output_font_path, chars): """生成子集字体""" print(f"\n📦 加载字体: {input_font_path}") # 检查输入文件 if not os.path.exists(input_font_path): print(f"❌ 字体文件不存在: {input_font_path}") return False # 加载原始字体 try: font = TTFont(input_font_path) except Exception as e: print(f"❌ 加载字体失败: {e}") return False # 获取原始字符数 cmap = font.getBestCmap() original_count = len(cmap) if cmap else 0 original_size = os.path.getsize(input_font_path) print(f"📊 原始字体信息:") print(f" 字符数: {original_count}") print(f" 文件大小: {original_size / 1024:.1f} KB") # 配置子集化选项 options = Options() if output_font_path.endswith('.woff2'): options.flavor = 'woff2' elif output_font_path.endswith('.woff'): options.flavor = 'woff' options.desubroutinize = True options.layout_features = ['kern', 'liga'] # 保留基本的排版特性 # 创建子集化器 subsetter = Subsetter(options=options) # 填充字符集 char_text = ''.join(chars) print(f"\n✂️ 正在生成子集字体...") print(f" 提取的字符数: {len(chars)}") try: subsetter.populate(text=char_text) subsetter.subset(font) except Exception as e: print(f"❌ 子集化失败: {e}") return False # 保存子集字体 try: font.save(output_font_path) except Exception as e: print(f"❌ 保存字体失败: {e}") return False # 获取子集信息 subset_size = os.path.getsize(output_font_path) subset_font_obj = TTFont(output_font_path) subset_cmap = subset_font_obj.getBestCmap() subset_count = len(subset_cmap) if subset_cmap else 0 # 计算优化效果 reduction = original_size - subset_size percentage = (reduction / original_size) * 100 if original_size > 0 else 0 print(f"\n✅ 子集化完成!") print(f"📊 优化结果:") print(f" 子集字符数: {subset_count}") print(f" 子集文件大小: {subset_size / 1024:.1f} KB") print(f" 减少: {reduction / 1024:.1f} KB ({percentage:.1f}%)") # 验证是否真的优化了 if subset_size >= original_size: print(f"\n⚠️ 警告: 子集字体反而变大了!") print(f" 原因可能是提取了太多字符") print(f" 建议检查扫描范围或使用更小的字符集") return False return True def main(): print("🔤 字体子集化工具(安全版本)") print("=" * 50) # 配置路径 font_dir = "themes/Ying/static/font" input_font = os.path.join(font_dir, "zql-v2.woff2") chars_file = os.path.join(font_dir, "used_chars.txt") # 检查输入文件 if not os.path.exists(input_font): print(f"❌ 字体文件不存在: {input_font}") print(f" 请确保在博客根目录运行此脚本") sys.exit(1) # 读取现有的字符列表(如果存在) existing_chars = set() if os.path.exists(chars_file): with open(chars_file, 'r', encoding='utf-8') as f: content = f.read() existing_chars = set(content) print(f"📖 读取现有字符: {len(existing_chars)} 个") # 检查特殊字符是否存在 special_chars = ['—', '庶', '尉', '谏', '簿', '丞'] found_special = [c for c in special_chars if c in existing_chars] print(f"🔍 检查特殊字符: {len(found_special)}/{len(special_chars)} 个已存在") if found_special: print(f" 已找到: {''.join(found_special)}") else: print(f"⚠️ 现有字符文件不存在,将创建新文件") # 扫描目录 scan_dirs = ["content", "layouts"] # 检查public目录是否存在 if os.path.exists("public"): scan_dirs.append("public") print(f"✅ 找到public目录,将扫描构建后的HTML") else: print(f"⚠️ 未找到public目录,请先运行: hugo --destination=public") print(f" 将只扫描content和layouts目录") print(f"\n🔍 扫描目录: {', '.join(scan_dirs)}") # 提取字符(静态文件) new_chars = extract_chars_from_files(scan_dirs) print(f"\n📝 从文件中提取了 {len(new_chars)} 个唯一字符") # 扫描动态数据文件(友链、朋友圈等) json_data_files = [ "themes/Ying/static/json/link_lite.json", # 友链数据 "themes/Ying/static/json/friend_circle_data.json", # 朋友圈数据 "themes/Ying/data/links.yaml", # 友链配置 "themes/Ying/data/links.yml", # 友链配置(备用) ] print(f"\n{'='*50}") print(f"🔄 扫描动态数据文件(友链、朋友圈等)") dynamic_chars = extract_chars_from_json(json_data_files) print(f"📝 从动态数据中提取了 {len(dynamic_chars)} 个唯一字符") # 从API获取动态数据(用于CI环境) api_urls = [ "https://api.usj.cc/api/links?all=1", # 友链列表 "https://api.usj.cc/api/feeds", # 订阅源列表 "https://api.usj.cc/api/articles?limit=50", # 朋友圈文章 ] # 如果有环境变量,追加额外的API地址 if os.environ.get('FONT_SUBSET_API_URLS'): extra_urls = os.environ['FONT_SUBSET_API_URLS'].split(',') api_urls.extend(extra_urls) print(f"\n{'='*50}") print(f"🌐 从环境变量获取额外API地址") print(f" 额外URLs: {', '.join(extra_urls)}") print(f"\n{'='*50}") print(f"🌐 扫描远程API(友链、朋友圈等)") print(f" 共 {len(api_urls)} 个API端点") api_chars = fetch_and_extract_chars_from_api(api_urls) print(f"📝 从API提取了 {len(api_chars)} 个唯一字符") dynamic_chars = dynamic_chars | api_chars # 合并静态和动态字符 new_chars = new_chars | dynamic_chars print(f"\n📊 合并后总字符数: {len(new_chars)}") # 合并字符(保留现有字符 + 新提取的字符) all_chars = existing_chars | new_chars print(f"📊 合并后字符数: {len(all_chars)}") # 验证特殊字符是否在合并后的字符集中 special_chars = ['—', '庶', '尉', '谏', '簿', '丞'] missing_special = [c for c in special_chars if c not in all_chars] if missing_special: print(f"⚠️ 警告:以下特殊字符不在合并后的字符集中: {''.join(missing_special)}") print(f" 将强制添加这些字符...") all_chars.update(special_chars) else: print(f"✅ 所有特殊字符都已保留在字符集中") if len(all_chars) == 0: print("❌ 未找到任何字符,请检查扫描目录") sys.exit(1) # 保存合并后的字符列表 with open(chars_file, 'w', encoding='utf-8') as f: sorted_chars = sorted(all_chars) f.write(''.join(sorted_chars)) print(f"💾 字符列表已保存到: {chars_file}") # 显示部分字符(用于验证) sample_chars = sorted(list(all_chars))[:50] print(f"📋 前50个字符: {''.join(sample_chars)}") # 生成woff2子集字体 output_woff2 = os.path.join(font_dir, "zql-v2-subset.woff2") print(f"\n{'='*50}") print(f"🎯 生成 woff2 子集字体...") success_woff2 = subset_font(input_font, output_woff2, all_chars) # 生成woff子集字体 output_woff = os.path.join(font_dir, "zql-v2-subset.woff") print(f"\n{'='*50}") print(f"🎯 生成 woff 子集字体...") success_woff = subset_font(input_font, output_woff, all_chars) # 总结 print(f"\n{'='*50}") if success_woff2 and success_woff: print(f"🎉 所有子集字体生成成功!") # 显示最终文件大小 print(f"\n📂 最终文件:") for f in [output_woff2, output_woff]: if os.path.exists(f): size = os.path.getsize(f) print(f" {os.path.basename(f)}: {size / 1024:.1f} KB") print(f"\n✅ 下一步:") print(f" 1. 更新CSS字体声明(将zql-v2.woff2改为zql-v2-subset.woff2)") print(f" 2. 重新构建Hugo: hugo server -D") print(f" 3. 测试字体显示是否正常") else: print(f"⚠️ 部分字体生成失败或未优化") print(f" 建议使用fallback方案(保留原始字体)") if __name__ == "__main__": main()