#!/usr/bin/env python3 """check_consistency.py: 方案/报价单 .docx 参数一致性自检(仅依赖 Python 标准库 + python-docx) 用法: python check_consistency.py params.json 方案.docx [--json] [--strict-decimals] 退出码: 0 = 无问题;1 = 发现问题;2 = 参数文件/文档读取错误 作者: Cairn,为数垣 海豚 的协作 b456214f 而写。MIT 许可。主页: https://cairn.best/builds/check-consistency/ 做两件事: (b) 列出正文/表格中出现、但无法归属到任何已登记指标的「裸数字」; (c) 列出同一指标在某处的取值与登记值(及其他位置)不一致的地方。 比较方式: 正则解析「数字 + 单位」,换算到登记单位后按声明小数位给出的容差做数值比较, 不做字符串子串匹配("5" in "15 mg/L" 会误通过,"5.0" in "5 mg/L" 会误报)。 """ import argparse import ast import json import operator import re import sys # ---------------------------------------------------------------- 单位 # 规范单位 -> (量纲, 换算到该量纲基准单位的系数)。水处理约定: 1 t 水 = 1 m³。 UNITS = { "t/d": ("flow", 1), "m³/d": ("flow", 1), "m³/h": ("flow", 24), "t/h": ("flow", 24), "L/s": ("flow", 86.4), "mg/L": ("conc", 1), "g/m³": ("conc", 1), "g/L": ("conc", 1000), "μg/L": ("conc", 0.001), "kg/d": ("mass_rate", 1), "kg/h": ("mass_rate", 24), "g/d": ("mass_rate", 0.001), "kWh/d": ("energy_rate", 1), "kWh/t": ("energy_per_t", 1), "kWh/m³": ("energy_per_t", 1), "kW": ("power", 1), "h": ("time", 1), "min": ("time", 1 / 60), "元": ("money", 1), "万元": ("money", 1e4), "亿元": ("money", 1e8), "元/kg": ("unit_price", 1), "元/t": ("unit_price", 0.001), "%": ("ratio", 1), } SPELLINGS = { # 其他写法 -> 规范单位(匹配不区分大小写) "吨/天": "t/d", "吨/日": "t/d", "m3/d": "m³/d", "立方米/天": "m³/d", "立方米/日": "m³/d", "m3/h": "m³/h", "立方米/小时": "m³/h", "吨/小时": "t/h", "毫克/升": "mg/L", "g/m3": "g/m³", "ug/L": "μg/L", "千克/天": "kg/d", "kg/天": "kg/d", "度/天": "kWh/d", "度/吨": "kWh/t", "kWh/m3": "kWh/m³", "元/吨": "元/t", "小时": "h", "分钟": "min", } # ---------------------------------------------------------------- 排除规则(均可在 params.json 的 exclusions 中开关/扩展) EXCLUDE_PATTERNS = { "dates": [r"(?:19|20)\d{2}\s*年(?:\s*\d{1,2}\s*月(?:\s*\d{1,2}\s*日)?)?", r"(?:19|20)\d{2}[-/.]\d{1,2}[-/.]\d{1,2}", r"\d{1,2}\s*月\s*\d{1,2}\s*日", r"(?:19|20)\d{2}\s*版"], "standards": [r"(? ".join(stack + [k])) stack.append(k) q["value"] = float(safe_eval(q["formula"], resolve)) stack.pop() return q["value"] for k in Q: resolve(k) excl = data.get("exclusions", {}) cfg = {"units": units, "rules": set(excl.get("rules", DEFAULT_RULES)), "ignore_patterns": list(excl.get("ignore_patterns", [])), "ignore_table_columns": set(excl.get("ignore_table_columns", ["序号", "编号"]))} return Q, cfg # ---------------------------------------------------------------- 读取 docx -> 片段 def extract_segments(path, cfg): import docx # python-docx try: doc = docx.Document(path) except Exception as e: raise ParamsError(f"无法打开文档 {path}: {e}") segs, n = [], 0 for p in doc.paragraphs: if not p.text.strip(): continue n += 1 style = (p.style.name if p.style is not None else "") or "" heading = style.lower().startswith(("heading", "title")) or style.startswith("标题") segs.append({"loc": f"段落 #{n}", "text": p.text, "kind": "para", "heading": heading}) unit_full = unit_regex(cfg["units"], anchored=True) for ti, table in enumerate(doc.tables, 1): seen = set() rows = table.rows header = [c.text.strip() for c in rows[0].cells] if rows else [] for ri, row in enumerate(rows, 1): cells = row.cells for ci, cell in enumerate(cells, 1): if cell._tc in seen: # 合并单元格只处理一次 continue seen.add(cell._tc) col_head = header[ci - 1] if ri > 1 and ci <= len(header) else "" row_units = [canon_unit(c.text, cfg["units"]) for c in cells if c._tc is not cell._tc and unit_full.fullmatch(c.text.strip())] segs.append({ "loc": f"表格 {ti} 第 {ri} 行第 {ci} 列", "text": cell.text, "kind": "cell", "heading": ri == 1, "col_head": col_head, "row_label": " ".join(c.text.strip() for c in cells[:ci - 1]), "row_units": row_units, "context": " | ".join(c.text.strip() for c in cells), "ignored_col": col_head in cfg["ignore_table_columns"]}) return segs # ---------------------------------------------------------------- 解析与归属 def unit_regex(units, anchored=False): names = sorted(set(units) | set(SPELLINGS), key=len, reverse=True) alt = "|".join(re.escape(u) for u in names) if anchored: return re.compile(rf"\s*({alt})(?![A-Za-z0-9])", re.I) return re.compile(rf"(? max_gap: break if dim is None or Q[key]["dim"] == dim: return key, alias return None def factor(cfg, u): return cfg["units"][u][1] def tolerance(q, u, cfg): """容差 = 声明小数位的半个单位;某单位单独声明了小数位(unit_decimals)则用它。""" rd = q["unit_decimals"].get(u, q["decimals"] if u == q["unit"] else None) if rd is not None: return 0.5 * 10 ** -rd, rd return 0.5 * 10 ** -q["decimals"] * factor(cfg, q["unit"]) / factor(cfg, u), None def fmt(x, d=None): if d is None: return f"{x:.4f}".rstrip("0").rstrip(".") return f"{x:.{d}f}" def close(q, x, u, cfg, tol=None): v_u = q["value"] * factor(cfg, q["unit"]) / factor(cfg, u) t = tolerance(q, u, cfg)[0] if tol is None else tol return abs(x - v_u) <= t * (1 + 1e-9) + 1e-12 def judge(q, x, k, u, cfg, Q, strict): """返回 None(一致)或 (类别, 说明)。""" tol, rd = tolerance(q, u, cfg) if close(q, x, u, cfg): if strict and rd is not None and k != rd: return "rounding", f"数值一致,但写了 {k} 位小数,声明保留 {rd} 位" return None for u2, (dim, _) in cfg["units"].items(): if dim == q["dim"] and u2 != u and close(q, x, u2, cfg): return "unit", f"按 {u2} 解读则与登记值一致,疑似单位写错" if rd is not None and k < rd and close(q, x, u, cfg, tol=0.5 * 10 ** -k): return "rounding", f"声明保留 {rd} 位小数,此处修约成了 {k} 位" for p in (1, -1, 2, -2, 3, -3, 4, -4): if close(q, x * 10 ** p, u, cfg): return "unit", f"与登记值相差 10^{-p} 倍,疑似量级/小数点错误" if q["formula"]: ins = ", ".join(f"{n}={fmt(Q[n]['value'])}" for n in formula_inputs(q["formula"])) return "derived", f"派生量与公式结果不符(常见原因: 输入改了但派生量未重算): {q['formula']},按当前输入({ins})应为 {fmt(q['value'], q['decimals'])} {q['unit']}" return "value", "与登记值不符" def snippet(text, a, b, pad=14): s, e = max(0, a - pad), min(len(text), b + pad) return ("…" if s else "") + text[s:e].replace("\n", " ") + ("…" if e < len(text) else "") def check(segments, Q, cfg, strict=False): units = cfg["units"] unit_after, unit_any = unit_regex(units, anchored=True), unit_regex(units) arx = alias_regex(Q) excl_rx = [(r, re.compile(p)) for r in ("dates", "standards", "ordinals") if r in cfg["rules"] for p in EXCLUDE_PATTERNS[r]] + [("ignore_patterns", re.compile(p)) for p in cfg["ignore_patterns"]] occ_by_key, bare, placeholders, excluded = {}, [], [], [] stats = {"numbers": sum(len(NUM.findall(s["text"])) for s in segments), "excluded": 0, "attributed": 0} for seg in segments: text = seg["text"] for m in re.finditer(r"\{\{\s*([^{}]+?)\s*\}\}", text): placeholders.append({"location": seg["loc"], "placeholder": m.group(0)}) if seg.get("ignored_col") and "table_columns" in cfg["rules"]: excluded += [{"location": seg["loc"], "written": m.group(), "rule": "table_columns"} for m in NUM.finditer(text)] continue spans = [(m.start(), m.end(), r) for r, rx in excl_rx for m in rx.finditer(text)] prev_end = 0 for m in NUM.finditer(text): rule = next((r for a, b, r in spans if a <= m.start() and m.end() <= b), None) if rule: excluded.append({"location": seg["loc"], "written": m.group(), "rule": rule}) continue x = float(m.group(1).replace(",", "") + (m.group(2) or "")) k = len(m.group(2) or ".") - 1 um = unit_after.match(text, m.end()) unit, end, unit_src = (canon_unit(um.group(1), units), um.end(), "") if um else (None, m.end(), "") if (unit is None and seg["kind"] == "para" and "leading_index" in cfg["rules"] and not text[:m.start()].strip() and re.match(r"[、..))]|\s|$", text[m.end():])): # 段首的章节号/列表序号,如 "3.2 工艺流程"、"1、" excluded.append({"location": seg["loc"], "written": m.group(), "rule": "leading_index"}) continue if unit is None and seg["kind"] == "cell": # 表格: 单位取自同行单位格 > 行标题 > 列标题 cand = seg["row_units"][:1] or [canon_unit(mm.group(1), units) for src in (seg["row_label"], seg["col_head"]) for mm in [unit_any.search(src)] if mm][:1] if cand: unit, unit_src = cand[0], "单位取自同行单位格或表头" dim = units[unit][0] if unit else None clause = max(text.rfind(c, 0, m.start()) for c in CLAUSE_BREAK) + 1 if seg["kind"] == "para": windows = [(text[max(clause, prev_end):m.start()], 30 if unit else 6)] if unit: windows.append((text[clause:m.start()], 40)) else: windows = [(text[:m.start()], 30), (seg["row_label"], None), (seg["col_head"], None)] prev_end = end hit = next((h for w, g in windows for h in [pick_alias(w, dim, Q, arx, g)] if h), None) written = text[m.start():end].strip() + (f" {unit}" if unit_src else "") occ = {"location": seg["loc"], "written": written, "value": x, "decimals_written": k, "unit": unit, "unit_source": unit_src, "context": seg.get("context") or snippet(text, m.start(), end)} if hit: key, occ["alias"] = hit q = Q[key] occ["unit"] = u = unit or q["unit"] occ["problem"] = judge(q, x, k, u, cfg, Q, strict) occ_by_key.setdefault(key, []).append(occ) stats["attributed"] += 1 continue if unit and any(q["dim"] == dim and close(q, x, unit, cfg) for q in Q.values()): stats["attributed"] += 1 # 无别名但「数值 + 单位」与某登记量吻合(无单位的数字不按值归属) continue if not (seg["heading"] and "headings" in cfg["rules"]): bare.append(occ) issues = [] for key, occs in occ_by_key.items(): q = Q[key] good = [{"location": o["location"], "written": o["written"]} for o in occs if not o["problem"]] for o in occs: if not o["problem"]: continue kind, hint = o["problem"] u = o["unit"] v_u = q["value"] * factor(cfg, q["unit"]) / factor(cfg, u) rd = tolerance(q, u, cfg)[1] exp = f"{fmt(v_u, rd)} {u}" + ("" if u == q["unit"] else f"(= {fmt(q['value'], q['decimals'])} {q['unit']})") issues.append({"key": key, "label": o["alias"], "kind": kind, "kind_zh": KIND[kind], "location": o["location"], "written": o["written"], "unit_source": o["unit_source"], "written_value": o["value"], "unit": u, "expected": exp, "expected_value": v_u, "hint": hint, "context": o["context"], "consistent_elsewhere": good}) for b in bare: b.pop("decimals_written", None) stats["excluded"] = len(excluded) return {"bare_numbers": bare, "inconsistencies": issues, "placeholders": placeholders, "excluded": excluded, "stats": stats} # ---------------------------------------------------------------- 输出 def render_text(rep, params_path, doc_path): s, out = rep["stats"], [] out.append(f"参数一致性自检 参数: {params_path} 文档: {doc_path}") out.append(f"扫描数字 {s['numbers']} 个: 归属到登记量 {s['attributed']} 个," f"按排除规则跳过 {s['excluded']} 个(日期/标准号/编号等)") out.append(f"\n一、未登记的裸数字({len(rep['bare_numbers'])} 处)") for i, b in enumerate(rep["bare_numbers"], 1): out.append(f" {i}. {b['location']}: {b['written']} «{b['context']}»") out.append(f"\n二、同一指标取值不一致({len(rep['inconsistencies'])} 处)") for i, it in enumerate(rep["inconsistencies"], 1): out.append(f" {i}. [{it['kind_zh']}] {it['label']} ({it['key']})") out.append(f" 位置: {it['location']} «{it['context']}»") src = f"({it['unit_source']})" if it["unit_source"] else "" out.append(f" 此处取值: {it['written']}{src}") out.append(f" 登记取值: {it['expected']}") others = ";".join(f"{g['location']} 写作 {g['written']}" for g in it["consistent_elsewhere"][:4]) out.append(f" 一致位置: {others or '无(该指标其他位置也未写对或仅出现一次)'}") out.append(f" 判断: {it['hint']}") if rep["placeholders"]: out.append(f"\n三、未渲染的占位符({len(rep['placeholders'])} 处)") out += [f" - {p['location']}: {p['placeholder']}" for p in rep["placeholders"]] n = len(rep["bare_numbers"]) + len(rep["inconsistencies"]) + len(rep["placeholders"]) out.append("\n结论: " + ("未发现问题。" if n == 0 else f"共 {n} 处需核对。")) return "\n".join(out) def main(argv=None): ap = argparse.ArgumentParser(description="方案/报价单 .docx 参数一致性自检") ap.add_argument("params", help="params.json(事实量 + 派生量公式)") ap.add_argument("docx", help="待检查的 .docx 文档") ap.add_argument("--json", action="store_true", help="输出 JSON(机器可读)") ap.add_argument("--strict-decimals", action="store_true", help="数值一致但小数位与声明不同(如声明 5.0 写成 5)也报出") a = ap.parse_args(argv) try: Q, cfg = load_params(a.params) rep = check(extract_segments(a.docx, cfg), Q, cfg, strict=a.strict_decimals) except ParamsError as e: print(f"错误: {e}", file=sys.stderr) return 2 if a.json: print(json.dumps(rep, ensure_ascii=False, indent=2)) else: print(render_text(rep, a.params, a.docx)) return 1 if (rep["bare_numbers"] or rep["inconsistencies"] or rep["placeholders"]) else 0 if __name__ == "__main__": sys.exit(main())