#!/usr/bin/env python3 """ contract_docx_lib.py — 合同修订核心库 固化验证通过的docx XML操作,不再每次重写。 用法: from contract_docx_lib import ContractEditor editor = ContractEditor("原文件.docx") editor.tracked_replace("原文片段", "新文片段") editor.add_clause("19.服务成果持续使用权", "条款内容...", after_clause=18) editor.renumber(19, 20) # 原19→20 errors = editor.validate() if not errors: editor.save("【修】原文件.docx") 关键操作顺序(renumber和新增条款): 1. 先做所有 tracked_replace(文本修改) 2. 再做 add_clause(新增子条款,如15.4) 3. 再做 renumber_range(先腾出编号空间) 4. 最后做 add_clause_before(插入新主条款,用已腾出的编号) 5. validate() 验证 6. save() 保存 """ import zipfile, io, copy, re, difflib from lxml import etree from datetime import datetime from pathlib import Path W = 'http://schemas.openxmlformats.org/wordprocessingml/2006/main' WP = 'http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing' XML_SPACE = '{http://www.w3.org/XML/1998/namespace}space' WNS = '{' + W + '}' def qn(tag): return f'{WNS}{tag}' def cjk_tokenize(text): """CJK每字一token,ASCII连续一token,标点单独token。 经验证的分词策略,不要改。""" tokens = [] i = 0 while i < len(text): ch = text[i] if '\u4e00' <= ch <= '\u9fff' or '\u3000' <= ch <= '\u303f' or ch in ',。、;:!?""''()【】《》—…·[]%%': tokens.append(ch) i += 1 elif ch.isascii() and ch.isalnum(): j = i while j < len(text) and text[j].isascii() and text[j].isalnum(): j += 1 tokens.append(text[i:j]) i = j else: tokens.append(ch) i += 1 return tokens class ContractEditor: """合同修订编辑器。一个实例对应一份合同文件。""" def __init__(self, filepath): self.filepath = Path(filepath) with open(filepath, 'rb') as f: self.original_bytes = f.read() with zipfile.ZipFile(io.BytesIO(self.original_bytes)) as z: self.doc_xml = z.read('word/document.xml') self.tree = etree.fromstring(self.doc_xml) self.body = self.tree.find(qn('body')) self._rev_id = 100 self._revision_date = datetime.now().strftime('%Y-%m-%dT%H:%M:%SZ') self._author = 'WB' self._rsid = '00AA0001' # 提取原文格式(核心:避免每次猜错格式) self._body_rpr = None # 正文格式(最常见的非加粗rPr) self._title_rpr = None # 条款标题格式(加粗的rPr) self._body_ppr = None self._extract_formats() def _extract_formats(self): """从原文提取正文和标题的rPr。 策略: - 正文格式:统计所有run的rPr,取出现最多的非加粗rPr - 标题格式:优先从条款编号标题段落(如"7.索赔条款")提取rPr, 而非简单取第一个加粗run(可能是合同大标题,字号不同) - 如果条款标题不加粗,标题格式回退到正文格式""" import re rpr_map = {} # serialized_rpr -> (count, rpr_element) clause_title_rpr = None # 从条款编号标题提取的格式 first_bold_rpr = None # 第一个加粗run的格式(fallback) for p in self.body.findall(qn('p')): # 获取段落全文,判断是否是条款编号标题(如 "7.索赔条款" "5.伴随服务") p_text = ''.join(t.text or '' for t in p.findall(f'.//{qn("t")}')).strip() is_clause_title = bool(re.match(r'^\d+[..、]\s*\S', p_text)) and len(p_text) < 30 for r in p.findall(qn('r')): rpr = r.find(qn('rPr')) txt = ''.join(t.text or '' for t in r.findall(qn('t'))) if not txt.strip() or len(txt) < 3: continue if rpr is not None: is_bold = rpr.find(qn('b')) is not None key = etree.tostring(rpr, encoding='unicode') if is_bold and first_bold_rpr is None: first_bold_rpr = rpr # 优先从条款标题段落提取标题格式 if is_clause_title and clause_title_rpr is None: clause_title_rpr = rpr if not is_bold: if key not in rpr_map: rpr_map[key] = [0, rpr] rpr_map[key][0] += 1 if self._body_ppr is None: ppr = p.find(qn('pPr')) txt = ''.join(t.text or '' for t in p.findall(f'.//{qn("t")}')) if ppr is not None and len(txt) > 20: self._body_ppr = ppr if rpr_map: best = max(rpr_map.values(), key=lambda x: x[0]) self._body_rpr = best[1] # 标题格式优先级:条款编号标题 > 第一个加粗run > 正文格式 self._title_rpr = clause_title_rpr or first_bold_rpr or self._body_rpr if self._title_rpr is None and self._body_rpr is not None: self._title_rpr = copy.deepcopy(self._body_rpr) etree.SubElement(self._title_rpr, qn('b')) def _next_id(self): self._rev_id += 1 return str(self._rev_id) def _mk_del(self, text, rpr=None): d = etree.Element(qn('del')) d.set(qn('id'), self._next_id()) d.set(qn('author'), self._author) d.set(qn('date'), self._revision_date) r = etree.SubElement(d, qn('r')) r.set(qn('rsidDel'), self._rsid) if rpr is not None: r.append(copy.deepcopy(rpr)) t = etree.SubElement(r, qn('delText')) t.set(XML_SPACE, 'preserve') t.text = text return d def _mk_ins(self, text, rpr=None): i = etree.Element(qn('ins')) i.set(qn('id'), self._next_id()) i.set(qn('author'), self._author) i.set(qn('date'), self._revision_date) r = etree.SubElement(i, qn('r')) r.set(qn('rsidR'), self._rsid) if rpr is not None: r.append(copy.deepcopy(rpr)) t = etree.SubElement(r, qn('t')) t.set(XML_SPACE, 'preserve') t.text = text return i def _mk_run(self, text, rpr=None): r = etree.Element(qn('r')) if rpr is not None: r.append(copy.deepcopy(rpr)) t = etree.SubElement(r, qn('t')) t.set(XML_SPACE, 'preserve') t.text = text return r def get_para_text(self, p): """获取段落的原始文本(不含删除标记中的文本)""" return ''.join(t.text or '' for t in p.findall(f'.//{qn("t")}')) def find_para(self, search_text): """查找包含指定文本的段落""" for p in self.body.findall(qn('p')): if search_text in self.get_para_text(p): return p return None def tracked_replace(self, old_text, new_text): """在整个文档中查找old_text并用修订模式替换为new_text。 使用字符级tokenizer+difflib实现精准修订。 返回True如果成功。""" for p in self.body.findall(qn('p')): runs = p.findall(f'.//{qn("r")}') if not runs: continue full = ''.join( ''.join(t.text or '' for t in r.findall(qn('t'))) for r in runs ) if old_text not in full: continue start = full.index(old_text) end = start + len(old_text) # 获取匹配位置的rPr rpr = None pos = 0 for r in runs: rt = ''.join(t.text or '' for t in r.findall(qn('t'))) if pos + len(rt) > start: rpr = r.find(qn('rPr')) break pos += len(rt) # 生成diff元素 if new_text == '': elems = [self._mk_del(old_text, rpr)] else: ot = cjk_tokenize(old_text) nt = cjk_tokenize(new_text) matcher = difflib.SequenceMatcher(None, ot, nt) elems = [] for tag, i1, i2, j1, j2 in matcher.get_opcodes(): if tag == 'equal': elems.append(self._mk_run(''.join(ot[i1:i2]), rpr)) elif tag == 'delete': elems.append(self._mk_del(''.join(ot[i1:i2]), rpr)) elif tag == 'insert': elems.append(self._mk_ins(''.join(nt[j1:j2]), rpr)) elif tag == 'replace': elems.append(self._mk_del(''.join(ot[i1:i2]), rpr)) elems.append(self._mk_ins(''.join(nt[j1:j2]), rpr)) # 定位受影响的runs并替换 pos = 0 first = last = None prefix_text = suffix_text = "" for idx, r in enumerate(runs): rt = ''.join(t.text or '' for t in r.findall(qn('t'))) run_end = pos + len(rt) if run_end > start and pos < end: if first is None: first = idx prefix_text = full[pos:start] last = idx suffix_text = full[end:run_end] if run_end > end else "" pos = run_end if first is None: continue ref = runs[first] # Find the actual paragraph (w:p) element to insert into para_elem = p # Determine insert position: find ref or its ancestor that is a direct child of p ref_ancestor = ref while ref_ancestor.getparent() is not para_elem and ref_ancestor.getparent() is not None: ref_ancestor = ref_ancestor.getparent() insert_pos = list(para_elem).index(ref_ancestor) # Remove runs (each from its own parent) for idx in range(last, first - 1, -1): r = runs[idx] r_parent = r.getparent() r_parent.remove(r) # If parent (e.g. w:ins) is now empty, remove it too if r_parent is not para_elem and len(r_parent) == 0: gp = r_parent.getparent() if gp is not None: gp.remove(r_parent) ip = insert_pos if prefix_text: para_elem.insert(ip, self._mk_run(prefix_text, rpr)) ip += 1 for e in elems: para_elem.insert(ip, e) ip += 1 if suffix_text: para_elem.insert(ip, self._mk_run(suffix_text, rpr)) return True return False def _get_leading_whitespace(self, para): """从段落中提取前导空格/tab模式。 很多中文文档的缩进不是通过w:ind实现的,而是通过文本中的空格字符。""" for r in para.findall(qn('r')): # Skip deleted runs if r.getparent().tag == qn('del'): continue for t in r.findall(qn('t')): if t.text: # Extract leading whitespace stripped = t.text.lstrip() if stripped: # Has actual content after whitespace return t.text[:len(t.text) - len(stripped)] elif t.text.isspace(): # Entire run is whitespace return t.text return '' def add_clause(self, full_text, after_search, use_title_format=False): """在包含after_search的段落之后插入新条款段落。 full_text: 新条款全文 after_search: 在包含此文本的段落之后插入 use_title_format: True=标题格式(加粗),False=正文格式 """ ref_para = self.find_para(after_search) if ref_para is None: return False rpr = self._title_rpr if use_title_format else self._body_rpr ppr = ref_para.find(qn('pPr')) or self._body_ppr # 复制相邻段落的前导空格模式 leading_ws = self._get_leading_whitespace(ref_para) if leading_ws and not full_text.startswith(leading_ws): full_text = leading_ws + full_text new_p = etree.Element(qn('p')) if ppr is not None: new_p.append(copy.deepcopy(ppr)) new_p.append(self._mk_ins(full_text, rpr)) idx = list(self.body).index(ref_para) self.body.insert(idx + 1, new_p) return True def add_clause_before(self, full_text, before_search, use_title_format=False): """在包含before_search的段落之前插入新条款段落。""" ref_para = self.find_para(before_search) if ref_para is None: return False rpr = self._title_rpr if use_title_format else self._body_rpr ppr = ref_para.find(qn('pPr')) or self._body_ppr # 复制相邻段落的前导空格模式 leading_ws = self._get_leading_whitespace(ref_para) if leading_ws and not full_text.startswith(leading_ws): full_text = leading_ws + full_text new_p = etree.Element(qn('p')) if ppr is not None: new_p.append(copy.deepcopy(ppr)) new_p.append(self._mk_ins(full_text, rpr)) idx = list(self.body).index(ref_para) self.body.insert(idx, new_p) return True def add_mixed_clause(self, title_text, content_text, after_search): """插入标题加粗+内容不加粗的新条款(两个段落)。 用于原文标题和内容分行的合同格式。""" ref_para = self.find_para(after_search) if ref_para is None: return False ppr = ref_para.find(qn('pPr')) or self._body_ppr idx = list(self.body).index(ref_para) # 复制相邻段落的前导空格模式 leading_ws = self._get_leading_whitespace(ref_para) if leading_ws: if not title_text.startswith(leading_ws): title_text = leading_ws + title_text if not content_text.startswith(leading_ws): content_text = leading_ws + content_text p_title = etree.Element(qn('p')) if ppr: p_title.append(copy.deepcopy(ppr)) p_title.append(self._mk_ins(title_text, self._title_rpr)) self.body.insert(idx + 1, p_title) p_content = etree.Element(qn('p')) if ppr: p_content.append(copy.deepcopy(ppr)) p_content.append(self._mk_ins(content_text, self._body_rpr)) self.body.insert(idx + 2, p_content) return True def renumber_clause(self, old_num, new_num): """把条款编号从old_num改为new_num(修订模式)。 从后往前扫描,避免重复修改。""" changed = 0 for p in reversed(self.body.findall(qn('p'))): runs = p.findall(f'.//{qn("r")}') for r in runs: for t in r.findall(qn('t')): if t.text and old_num in t.text: rpr_e = r.find(qn('rPr')) parent = r.getparent() idx_r = list(parent).index(r) pos = t.text.index(old_num) prefix = t.text[:pos] suffix = t.text[pos + len(old_num):] parent.remove(r) ip = idx_r if prefix: parent.insert(ip, self._mk_run(prefix, rpr_e)) ip += 1 parent.insert(ip, self._mk_del(old_num, rpr_e)) ip += 1 parent.insert(ip, self._mk_ins(new_num, rpr_e)) ip += 1 if suffix: parent.insert(ip, self._mk_run(suffix, rpr_e)) changed += 1 break return changed def renumber_range(self, start, shift=1): """从start开始,所有现有条款编号+shift。从后往前处理。 注意:先调用此方法腾出编号空间,再插入新条款。 例:要在18后插入新19条: editor.renumber_range(19, 1) # 19→20, 20→21, 21→22 editor.add_clause_before("19.新条款内容", before_search="20.合同生效") """ max_num = 0 for p in self.body.findall(qn('p')): txt = self.get_para_text(p) for m in re.finditer(r'(\d+)[..]', txt): n = int(m.group(1)) if n > max_num: max_num = n for n in range(max_num, start - 1, -1): self.renumber_clause(f'{n}.', f'{n + shift}.') self.renumber_clause(f'{n}.', f'{n + shift}.') def renumber_chinese(self, old_cn, new_cn): """中文编号顺延,如 "第十三条" → "第十四条"。""" return self.renumber_clause(old_cn, new_cn) def validate(self): """交付前验证。返回错误列表,空列表=通过。""" errors = [] # 1. 编号连续性 clause_nums = [] for p in self.body.findall(qn('p')): accepted = '' for child in p: tag = child.tag.split('}')[-1] if '}' in child.tag else child.tag if tag == 'r': accepted += ''.join(t.text or '' for t in child.findall(qn('t'))) elif tag == 'ins': accepted += ''.join(t.text or '' for t in child.findall(f'.//{qn("t")}')) m = re.match(r'^(\d+)[..]', accepted.strip()) if m: clause_nums.append(int(m.group(1))) main_clauses = sorted(set(clause_nums)) for i in range(1, len(main_clauses)): if main_clauses[i] - main_clauses[i-1] > 1: errors.append(f"编号跳跃: {main_clauses[i-1]}→{main_clauses[i]},缺少{main_clauses[i-1]+1}") # 2. 字号一致性(WB的ins内容 vs 原文正文) if self._body_rpr is not None: body_sz = None sz_elem = self._body_rpr.find(qn('sz')) if sz_elem is not None: body_sz = sz_elem.get(qn('val')) if body_sz: for ins in self.tree.findall(f'.//{qn("ins")}'): if ins.get(qn('author')) != self._author: continue for r in ins.findall(qn('r')): rpr = r.find(qn('rPr')) txt = ''.join(t.text or '' for t in r.findall(qn('t'))) if not txt.strip(): continue if rpr is not None: ins_sz = rpr.find(qn('sz')) if ins_sz is not None: val = ins_sz.get(qn('val')) is_bold = rpr.find(qn('b')) is not None if val != body_sz and not is_bold: errors.append(f"字号不一致: ins sz={val} vs 原文sz={body_sz},'{txt[:30]}'") # 3. 加粗规则(内容不应加粗) for ins in self.tree.findall(f'.//{qn("ins")}'): if ins.get(qn('author')) != self._author: continue for r in ins.findall(qn('r')): rpr = r.find(qn('rPr')) txt = ''.join(t.text or '' for t in r.findall(qn('t'))) if not txt.strip() or len(txt.strip()) < 5: continue is_bold = rpr is not None and rpr.find(qn('b')) is not None is_clause_title = bool(re.match(r'^\d+[..]\S', txt.strip())) or bool(re.match(r'^第.{1,3}条', txt.strip())) or bool(re.match(r'^[一二三四五六七八九十]{1,3}、', txt.strip())) if is_bold and not is_clause_title: errors.append(f"不应加粗: '{txt[:40]}'") return errors def dump_numbering(self): """输出accepted view的编号序列,用于人工确认""" result = [] for p in self.body.findall(qn('p')): accepted = '' for child in p: tag = child.tag.split('}')[-1] if '}' in child.tag else child.tag if tag == 'r': accepted += ''.join(t.text or '' for t in child.findall(qn('t'))) elif tag == 'ins': accepted += ''.join(t.text or '' for t in child.findall(f'.//{qn("t")}')) m = re.match(r'^(\d+)[..]', accepted.strip()) if m: result.append(f"{m.group(1)}. {accepted.strip()[:60]}") return result def save(self, output_path): """保存修订后的文件""" new_doc_xml = etree.tostring(self.tree, xml_declaration=True, encoding='UTF-8', standalone=True) with zipfile.ZipFile(io.BytesIO(self.original_bytes)) as z: settings = z.read('word/settings.xml') stree = etree.fromstring(settings) if stree.find(f'.//{qn("trackRevisions")}') is None: stree.append(etree.Element(qn('trackRevisions'))) new_settings = etree.tostring(stree, xml_declaration=True, encoding='UTF-8', standalone=True) buf = io.BytesIO() with zipfile.ZipFile(io.BytesIO(self.original_bytes)) as zin: with zipfile.ZipFile(buf, 'w', zipfile.ZIP_DEFLATED) as zout: for item in zin.infolist(): if item.filename == 'word/document.xml': zout.writestr(item, new_doc_xml) elif item.filename == 'word/settings.xml': zout.writestr(item, new_settings) else: zout.writestr(item, zin.read(item.filename)) with open(output_path, 'wb') as f: f.write(buf.getvalue()) return output_path class ZhujiajaoOpinion: """朱家角审查意见表格填写器。严格使用模板结构,不自创格式。""" TEMPLATE_PATH = Path.home() / ".hermes/shared/模版库/朱家角 审查意见【模板】.docx" def __init__(self, template_path=None): tpath = Path(template_path) if template_path else self.TEMPLATE_PATH with open(tpath, 'rb') as f: self.tmpl_bytes = f.read() with zipfile.ZipFile(io.BytesIO(self.tmpl_bytes)) as z: self.doc_xml = z.read('word/document.xml') self.tree = etree.fromstring(self.doc_xml) self.body = self.tree.find(qn('body')) def fill(self, contract_name, items, has_modifications=True): """填写审查意见。 contract_name: 合同名称(填入标题《》中间) items: [(条文位置, 原文, 修订后), ...] has_modifications: False则保留"无法律修改意见" """ # 1. 填标题——找到空格run替换 for p in self.body.findall(qn('p')): runs = p.findall(f'.//{qn("r")}') for r in runs: for t in r.findall(qn('t')): if t.text and t.text.strip() == '' and len(t.text) >= 2: parent_txt = ''.join( tt.text or '' for rr in runs for tt in rr.findall(qn('t')) ) if '关于《' in parent_txt: t.text = contract_name # 2. 处理"无法律修改意见" if has_modifications: for p in self.body.findall(qn('p')): txt = ''.join(t.text or '' for t in p.findall(f'.//{qn("t")}')) if '无法律修改意见' in txt: for r in p.findall(f'.//{qn("r")}'): for t in r.findall(qn('t')): if '无法律修改意见' in (t.text or ''): t.text = '' # 3. 填表格 if not items: return tbl = self.body.find(qn('tbl')) if tbl is None: return rows = tbl.findall(qn('tr')) # Row 0 = header, Row 1+ = data rows # 获取表头rPr header_rpr = None for hc in rows[0].findall(qn('tc')): for hr in hc.findall(f'.//{qn("r")}'): rr = hr.find(qn('rPr')) if rr: header_rpr = rr break if header_rpr: break # 确保有足够数据行 template_row = rows[1] if len(rows) > 1 else None while len(tbl.findall(qn('tr'))) - 1 < len(items): if template_row is not None: tbl.append(copy.deepcopy(template_row)) rows = tbl.findall(qn('tr')) # 填写数据 for i, (clause, orig_text, modified_text) in enumerate(items): if i + 1 >= len(rows): break row = rows[i + 1] cells = row.findall(qn('tc')) if len(cells) < 3: continue for ci, text in enumerate([clause, orig_text, modified_text]): cell = cells[ci] p = cell.find(qn('p')) if p is None: p = etree.SubElement(cell, qn('p')) for r in p.findall(qn('r')): p.remove(r) r = etree.SubElement(p, qn('r')) if header_rpr: new_rpr = copy.deepcopy(header_rpr) b = new_rpr.find(qn('b')) if b is not None: new_rpr.remove(b) if '注:' in text: color = new_rpr.find(qn('color')) if color is None: color = etree.SubElement(new_rpr, qn('color')) color.set(qn('val'), 'FF0000') r.append(new_rpr) t = etree.SubElement(r, qn('t')) t.set(XML_SPACE, 'preserve') t.text = text # 删除多余空行 rows = tbl.findall(qn('tr')) for i in range(len(rows) - 1, len(items), -1): tbl.remove(rows[i]) def save(self, output_path): new_doc = etree.tostring(self.tree, xml_declaration=True, encoding='UTF-8', standalone=True) buf = io.BytesIO() with zipfile.ZipFile(io.BytesIO(self.tmpl_bytes)) as zin: with zipfile.ZipFile(buf, 'w', zipfile.ZIP_DEFLATED) as zout: for item in zin.infolist(): if item.filename == 'word/document.xml': zout.writestr(item, new_doc) else: zout.writestr(item, zin.read(item.filename)) with open(output_path, 'wb') as f: f.write(buf.getvalue()) return output_path