feat: export core Hermes skills

This commit is contained in:
2026-07-15 02:45:56 +00:00
parent a028b63eda
commit 54711fee2a
308 changed files with 41310 additions and 1 deletions
@@ -0,0 +1,136 @@
#!/usr/bin/env python3
"""Verify WB INS font consistency against same-paragraph original runs.
Usage: python wb-ins-font-verify.py <docx_path>
Document-agnostic: doesn't hardcode font names — compares each WB INS run
against the nearest original (non-tracked) run in the same paragraph.
Checks: rFonts (eastAsia, ascii), w:sz, w:hint, and bold consistency.
Exit code 0 = pass, 1 = issues found.
"""
import sys, zipfile, io
from lxml import etree
def qn(tag):
return '{http://schemas.openxmlformats.org/wordprocessingml/2006/main}' + tag
def get_rpr_info(rpr):
if rpr is None:
return {'ea': None, 'ascii': None, 'hint': None, 'sz': None, 'bold': False}
rfonts = rpr.find(qn('rFonts'))
sz = rpr.find(qn('sz'))
b = rpr.find(qn('b'))
return {
'ea': rfonts.get(qn('eastAsia')) if rfonts is not None else None,
'ascii': rfonts.get(qn('ascii')) if rfonts is not None else None,
'hint': rfonts.get(qn('hint')) if rfonts is not None else None,
'sz': sz.get(qn('val')) if sz is not None else None,
'bold': b is not None,
}
def main(docx_path):
with zipfile.ZipFile(io.BytesIO(open(docx_path, 'rb').read())) as z:
tree = etree.fromstring(z.read('word/document.xml'))
body = tree.find(qn('body'))
issues = []
total = 0
# Use recursive search to find ALL paragraphs, including those inside tables.
# Many Chinese contracts (esp. government templates) nest body text inside w:tbl.
# body.findall(qn('p')) only gets direct children and misses table content entirely.
for pi, p in enumerate(body.findall('.//' + qn('p'))):
# Collect WB INS runs
wb_runs = []
for ins in p.findall('.//' + qn('ins')):
if ins.get(qn('author')) != 'WB':
continue
for r in ins.findall(qn('r')):
t = r.find(qn('t'))
if t is not None and (t.text or '').strip():
wb_runs.append((t.text, r))
if not wb_runs:
continue
# Collect original (non-tracked) runs in same paragraph
orig_info = None
for r in p.findall(qn('r')):
parent = r.getparent()
if parent.tag in [qn('ins'), qn('del')]:
continue
t = r.find(qn('t'))
if t is not None and (t.text or '').strip():
orig_info = get_rpr_info(r.find(qn('rPr')))
break # first non-trivial original run
for text, r in wb_runs:
total += 1
wb_info = get_rpr_info(r.find(qn('rPr')))
short = text[:50]
# PRIMARY STANDARD: WB INS run must match the same-paragraph original run.
# Do NOT impose an absolute "must have explicit eastAsia/ascii" rule — many
# Chinese government templates (e.g. 教育部 GF-2021 校外培训合同) define CJK
# fonts via hint="eastAsia"+cs WITHOUT explicit eastAsia/ascii attrs. A correctly
# inherited single-char replacement in such a doc has ea=None/ascii=None and is
# CORRECT — flagging it "MISSING FONT" is a false positive (2026-06-17 教训).
if orig_info is not None:
# Compare against original: ea, ascii, hint, sz must all match the orig run.
for key, label in [('ea','eastAsia'),('ascii','ascii'),('hint','hint'),('sz','sz')]:
if wb_info[key] != orig_info[key]:
issues.append(f"P{pi} {label.upper()} MISMATCH vs同段原文: '{short}' wb={wb_info[key]} orig={orig_info[key]}")
else:
# No original run to compare (fully-new paragraph). Require hint present
# (CJK safety) but don't hard-require explicit ea/ascii.
if not wb_info['hint']:
issues.append(f"P{pi} MISSING HINT (无同段原文可比): '{short}'")
# Phase 2: Title bold consistency check
# Collect all "第X条" title patterns and verify bold consistency
import re
title_bolds = {} # paragraph_index -> bold status of "第X条" text
for pi, p in enumerate(body.findall(qn('p'))):
for elem in p.iter():
if elem.tag == qn('t') and elem.text:
if re.match(r'^第[一二三四五六七八九十百千\d]+条', elem.text.strip()):
# Find parent run's bold status
run = elem.getparent()
if run is not None and run.tag == qn('r'):
rpr = run.find(qn('rPr'))
bold = rpr.find(qn('b')) is not None if rpr is not None else False
title_bolds[pi] = bold
elif run is not None and run.tag == qn('ins'):
# Inside w:ins — check the r inside
pass
# Also check inside ins elements
if elem.tag == qn('ins') and elem.get(qn('author')) == 'WB':
for r in elem.findall(qn('r')):
t = r.find(qn('t'))
if t is not None and t.text and re.match(r'^第[一二三四五六七八九十百千\d]+条', t.text.strip()):
rpr = r.find(qn('rPr'))
bold = rpr.find(qn('b')) is not None if rpr is not None else False
title_bolds[pi] = bold
if title_bolds:
bold_values = list(title_bolds.values())
majority_bold = bold_values.count(True) > bold_values.count(False)
for pi, is_bold in title_bolds.items():
if is_bold != majority_bold:
issues.append(f"P{pi} TITLE BOLD INCONSISTENT: bold={is_bold}, majority={majority_bold}")
if issues:
print(f"FAIL: {len(issues)} issues in {total} WB INS runs")
for i in issues:
print(f" {i}")
return 1
else:
print(f"PASS: all {total} WB INS runs font-consistent, {len(title_bolds)} title(s) bold-consistent")
return 0
if __name__ == '__main__':
if len(sys.argv) != 2:
print(f"Usage: {sys.argv[0]} <docx_path>")
sys.exit(2)
sys.exit(main(sys.argv[1]))