This commit is contained in:
Vaica committed 2026-06-24 23:08:35 +08:00
1 parent 450a28c81d
commit 0be8aa9668
7 files changed
+588 -203

No files matched your search

+127 -1
View File
@@ -50,6 +50,92 @@ def extract_chars_from_files(directories, extensions=('.html', '.md', '.css')):
return chars
def extract_chars_from_json(json_files):
"""从JSON数据文件中提取字符(用于友链、朋友圈等动态内容)"""
chars = set()
import json
for json_file in json_files:
if not os.path.exists(json_file):
print(f"⚠️ JSON文件不存在,跳过: {json_file}")
continue
print(f"🔍 扫描JSON文件: {json_file}")
try:
with open(json_file, 'r', encoding='utf-8') as f:
data = json.load(f)
# 将整个JSON转换为字符串,提取所有字符
content = json.dumps(data, ensure_ascii=False)
# 提取中文字符
chinese_chars = re.findall(r'[一-鿿]', content)
chars.update(chinese_chars)
# 提取中文标点
cjk_punct = re.findall(r'[ -〿＀-￯]', content)
chars.update(cjk_punct)
# 提取英文和数字
ascii_chars = re.findall(r'[a-zA-Z0-9]', content)
chars.update(ascii_chars)
print(f" ✅ 从 {json_file} 提取了 {len(chinese_chars)} 个中文字符")
except Exception as e:
print(f"⚠️ 读取JSON文件失败 {json_file}: {e}")
return chars
def fetch_and_extract_chars_from_api(urls):
"""从API接口获取数据并提取字符(用于动态内容如友链、朋友圈)"""
chars = set()
try:
import urllib.request
import json
import ssl
# 创建不验证SSL的上下文(用于内网或自签名证书)
context = ssl._create_unverified_context()
for url in urls:
print(f"🌐 尝试从API获取: {url}")
try:
req = urllib.request.Request(
url,
headers={
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
}
)
with urllib.request.urlopen(req, context=context, timeout=15) as response:
data = response.read().decode('utf-8')
try:
json_data = json.loads(data)
content = json.dumps(json_data, ensure_ascii=False)
# 提取中文字符
chinese_chars = re.findall(r'[一-鿿]', content)
chars.update(chinese_chars)
print(f" ✅ 从API提取了 {len(chinese_chars)} 个中文字符")
except json.JSONDecodeError:
print(f" ⚠️ 响应不是有效的JSON,跳过")
except urllib.error.URLError as e:
print(f" ❌ API请求失败: {e}")
except Exception as e:
print(f" ❌ 处理API响应失败: {e}")
except ImportError:
print("⚠️ urllib不可用,跳过API扫描")
return chars
def subset_font(input_font_path, output_font_path, chars):
"""生成子集字体"""
print(f"\n📦 加载字体: {input_font_path}")
@@ -177,10 +263,50 @@ def main():
print(f"\n🔍 扫描目录: {', '.join(scan_dirs)}")
# 提取字符
# 提取字符(静态文件)
new_chars = extract_chars_from_files(scan_dirs)
print(f"\n📝 从文件中提取了 {len(new_chars)} 个唯一字符")
# 扫描动态数据文件(友链、朋友圈等)
json_data_files = [
"themes/Ying/static/json/link_lite.json", # 友链数据
"themes/Ying/static/json/friend_circle_data.json", # 朋友圈数据
"themes/Ying/data/links.yaml", # 友链配置
"themes/Ying/data/links.yml", # 友链配置(备用)
]
print(f"\n{'='*50}")
print(f"🔄 扫描动态数据文件(友链、朋友圈等)")
dynamic_chars = extract_chars_from_json(json_data_files)
print(f"📝 从动态数据中提取了 {len(dynamic_chars)} 个唯一字符")
# 从API获取动态数据(用于CI环境)
api_urls = [
"https://api.usj.cc/api/links?all=1", # 友链列表
"https://api.usj.cc/api/feeds", # 订阅源列表
"https://api.usj.cc/api/articles?limit=50", # 朋友圈文章
]
# 如果有环境变量,追加额外的API地址
if os.environ.get('FONT_SUBSET_API_URLS'):
extra_urls = os.environ['FONT_SUBSET_API_URLS'].split(',')
api_urls.extend(extra_urls)
print(f"\n{'='*50}")
print(f"🌐 从环境变量获取额外API地址")
print(f" 额外URLs: {', '.join(extra_urls)}")
print(f"\n{'='*50}")
print(f"🌐 扫描远程API(友链、朋友圈等)")
print(f" 共 {len(api_urls)} 个API端点")
api_chars = fetch_and_extract_chars_from_api(api_urls)
print(f"📝 从API提取了 {len(api_chars)} 个唯一字符")
dynamic_chars = dynamic_chars | api_chars
# 合并静态和动态字符
new_chars = new_chars | dynamic_chars
print(f"\n📊 合并后总字符数: {len(new_chars)}")
# 合并字符(保留现有字符 + 新提取的字符)
all_chars = existing_chars | new_chars
print(f"📊 合并后字符数: {len(all_chars)}")