delivery_docs_build.py 35 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. r"""交付文档构建器:`docs/src/*.md` → `docs/*.docx`(用户令 2026-09-22)。
  4. ## 为什么要有它(而不是拿 Word 手排)
  5. 用户令要求三份文档「区分章节目录、文表图并茂、字体字号分类统一」。手排的三份文档**必然**字体字号漂移、
  6. 图表编号对不上、改一处忘一处。所以内容用 markdown 写(可 diff、进版本库),排版由本器**一套样式表**统一执行:
  7. | 元素 | 字体 | 字号 | 对齐 / 缩进 |
  8. |---|---|---|---|
  9. | 封面标题 | 黑体 | 小一(26pt) | 居中 |
  10. | 章标题(一级 `##`) | 黑体 | 三号(16pt) | 左对齐,**章前分页** |
  11. | 节标题(二级 `###`) | 黑体 | 四号(14pt) | 左对齐 |
  12. | 小节标题(三级 `####`) | 黑体 | 小四(12pt) 加粗 | 左对齐 |
  13. | 正文 | 宋体 | 小四(12pt) | 首行缩进 2 字符、1.5 倍行距、两端对齐 |
  14. | 列表 | 宋体 | 小四(12pt) | 悬挂缩进 |
  15. | 表题(表上方) | 黑体 | 五号(10.5pt) | 居中 |
  16. | 表头 | 黑体 | 五号(10.5pt) 加粗 | 居中 + 浅灰底纹 + 跨页重复 |
  17. | 表格正文 | 宋体 | 五号(10.5pt)(≥6 列降 9pt) | 左对齐 |
  18. | 图(居中)与图题(图下方) | 黑体 | 五号(10.5pt) | 居中 |
  19. | 页眉 / 页脚页码 | 宋体 | 小五(9pt) | 居中 |
  20. 英文与数字用 Times New Roman(中式公文惯例:中文宋体/黑体 + 西文 Times New Roman)。
  21. ## markdown 约定(渲染器支持的子集,写手必须照此写)
  22. # 文档标题 (只此一处,进封面)
  23. ## 1 章 / ### 1.1 节 / #### 1.1.1 小节
  24. 普通段落(一行一段,段间空行)
  25. - 无序项 1. 有序项
  26. | 表头 | 表头 | + |---| 分隔行 → 表格
  27. **表 3-1 标题**(紧挨表格上方一行) → 表题
  28. ![图 3-1 标题](figures/xxx.png) → 图片 + 图题
  29. > 引用段
  30. --- → 忽略
  31. ## 用法
  32. python scripts/delivery_docs_build.py # 三份全渲染
  33. python scripts/delivery_docs_build.py --only req # 只渲染一份 (req/des/dat)
  34. python scripts/delivery_docs_build.py --check # 只体检源件(章/节/表/图/字数、图是否在位)
  35. python scripts/delivery_docs_build.py --spec # 打印本器的排版规范表
  36. 退出码: 0 全部成功 · 5 有源件缺件/图缺件(逐条打印,仍会尽力渲染其余内容)
  37. """
  38. from __future__ import annotations
  39. try:
  40. from app_common.app_common_guanlan.api import install_root as _install_root
  41. except ImportError: # 理论不可达;包结构异常时回退到按位置上跳
  42. from pathlib import Path as _P
  43. def _install_root(_f): return _P(_f).resolve().parents[4]
  44. import argparse
  45. import pathlib
  46. import re
  47. import sys
  48. ROOT = _install_root(__file__)
  49. from docx import Document # noqa: E402
  50. from docx.enum.section import WD_SECTION # noqa: E402
  51. from docx.enum.table import WD_TABLE_ALIGNMENT # noqa: E402
  52. from docx.enum.text import WD_ALIGN_PARAGRAPH, WD_LINE_SPACING # noqa: E402
  53. from docx.oxml import OxmlElement # noqa: E402
  54. from docx.oxml.ns import qn # noqa: E402
  55. from docx.shared import Cm, Pt, RGBColor # noqa: E402
  56. VERSION = __import__('importlib').import_module('src.version').VERSION
  57. HEI, SONG, WEST = '黑体', '宋体', 'Times New Roman'
  58. SZ = dict(cover=26, h1=16, h2=14, h3=12, body=12, table=10.5, table_small=9, caption=10.5, header=9)
  59. DOCS = {
  60. 'req': dict(src='docs/src/需求分析_观澜_{v}.md', out='docs/需求分析_观澜_{v}.docx',
  61. title='需求分析', subtitle='观澜 v2 · 风电场智能分析系统'),
  62. 'des': dict(src='docs/src/系统设计说明_观澜_{v}.md', out='docs/系统设计说明_观澜_{v}.docx',
  63. title='系统设计说明', subtitle='观澜 v2 · 风电场智能分析系统'),
  64. 'dat': dict(src='docs/src/数据要求说明_观澜_{v}.md', out='docs/数据要求说明_观澜_{v}.docx',
  65. title='数据要求说明',
  66. subtitle='观澜 v2 · 风电场智能分析系统(结合现场《数据分析收资要求-v3》)',
  67. # ★2026-09-22 用户令: 这份是**纯数据需求规格** —— 不体现样本场接入情况, 也不体现实现细节。
  68. # 下面这组模式由 `--check`/`--verify` 机器扫描, 命中即 FAIL(不靠人自觉)。
  69. strict_terms=True),
  70. }
  71. # 数据要求说明的"不得出现"模式(用户令 2026-09-22):
  72. # 英文测点名称 / 文件名称 / 落盘位置 / 机组名称编号 / 样本场现状字样。
  73. # 中文业务含义名称与「中文(英文简写)」(如 平均无故障间隔(MTBF))不在禁止之列。
  74. STRICT_PATTERNS = (
  75. ('英文测点标识(snake_case)', r'\b[a-z][a-z0-9]*_[a-z0-9_]+\b'),
  76. ('文件扩展名', r'\.(csv|parquet|json|yaml|yml|xlsx|xls|pdf|mdb|xml|docx|doc)\b'),
  77. ('目录/落盘位置', r'(data/raw|outputs/|configs/|scripts/|reference/|docs/|data\\)'),
  78. ('机组名称/编号', r'(WTG\d+|\b\d{2}[A-Z]\b)'),
  79. ('样本场现状字样', r'(已在位|未到位|现状|实测|样本场|本场|催缴)'),
  80. )
  81. # ★去标识化词表(用户令 2026-09-22「内容参考如东风电场,但不体现如东风电场」):
  82. # 三份文档是**对外**交付件, 不得出现样本场的中文名、罗马化形态、业主、地域、OEM 与第三方机构名。
  83. # 词表放在代码里 ⇒ `--check`/`--verify` 与 `guanlan.py check` 都会机器扫描, 不靠人自觉。
  84. # ★2026-09-22 补(数据要求说明起草时实逮): 样本场 CMS 导出包名是 `CMS_RuDong_CGN_202603-04`
  85. # —— **大小写混合**的罗马化场名/业主缩写能绕过精确匹配,所以匹配一律**大小写不敏感**(见 forbidden_hits)。
  86. FORBIDDEN = ('如东', '如海', '广核', '江苏', 'rudong', 'cgn',
  87. '西门子', '上海电气', 'swt', '大生', '4.0-130')
  88. SPEC_ROWS = [
  89. ('封面标题', '黑体', '小一(26pt)', '居中'),
  90. ('章标题(一级)', '黑体', '三号(16pt)', '左对齐,章前分页'),
  91. ('节标题(二级)', '黑体', '四号(14pt)', '左对齐'),
  92. ('小节标题(三级)', '黑体', '小四(12pt)加粗', '左对齐'),
  93. ('正文', '宋体(西文 Times New Roman)', '小四(12pt)', '首行缩进 2 字符,1.5 倍行距,两端对齐'),
  94. ('列表', '宋体', '小四(12pt)', '悬挂缩进 0.74 cm'),
  95. ('表题', '黑体', '五号(10.5pt)', '居中,置于表格上方'),
  96. ('表头', '黑体', '五号(10.5pt)加粗', '居中,浅灰底纹,跨页重复'),
  97. ('表格正文', '宋体', '五号(10.5pt);≥6 列降 9pt', '左对齐,垂直居中'),
  98. ('图', '—', '宽度 ≤ 15 cm', '居中'),
  99. ('图题', '黑体', '五号(10.5pt)', '居中,置于图下方'),
  100. ('页眉', '宋体', '小五(9pt)', '居中'),
  101. ('页脚页码', '宋体', '小五(9pt)', '居中(第 X 页 / 共 Y 页)'),
  102. ]
  103. WARN: list[str] = []
  104. # ── 低层 docx 工具 ─────────────────────────────────────────────────────────────
  105. def _rfonts(el, ascii_font: str, ea_font: str):
  106. rpr = el.get_or_add_rPr()
  107. rf = rpr.find(qn('w:rFonts'))
  108. if rf is None:
  109. rf = OxmlElement('w:rFonts')
  110. rpr.append(rf)
  111. rf.set(qn('w:ascii'), ascii_font)
  112. rf.set(qn('w:hAnsi'), ascii_font)
  113. rf.set(qn('w:eastAsia'), ea_font)
  114. def set_run(run, size: float, *, ea: str = SONG, ascii_font: str = WEST, bold: bool = False):
  115. run.font.size = Pt(size)
  116. run.font.bold = bold
  117. run.font.name = ascii_font
  118. run.font.color.rgb = RGBColor(0, 0, 0)
  119. _rfonts(run._element, ascii_font, ea)
  120. def style_setup(doc: Document):
  121. """把内置样式(Heading 1-4 / Normal / 表格)改成文档的统一口径。"""
  122. normal = doc.styles['Normal']
  123. normal.font.name = WEST
  124. normal.font.size = Pt(SZ['body'])
  125. _rfonts(normal.element, WEST, SONG)
  126. normal.paragraph_format.line_spacing_rule = WD_LINE_SPACING.MULTIPLE
  127. normal.paragraph_format.line_spacing = 1.5
  128. normal.paragraph_format.space_after = Pt(0)
  129. spec = {'Heading 1': (SZ['h1'], True), 'Heading 2': (SZ['h2'], True),
  130. 'Heading 3': (SZ['h3'], True), 'Heading 4': (SZ['h3'], True)}
  131. for name, (size, bold) in spec.items():
  132. st = doc.styles[name]
  133. st.font.name = WEST
  134. st.font.size = Pt(size)
  135. st.font.bold = bold
  136. st.font.italic = False
  137. st.font.color.rgb = RGBColor(0, 0, 0)
  138. _rfonts(st.element, WEST, HEI)
  139. pf = st.paragraph_format
  140. pf.line_spacing_rule = WD_LINE_SPACING.MULTIPLE
  141. pf.line_spacing = 1.5
  142. pf.space_before = Pt(12 if name != 'Heading 1' else 0)
  143. pf.space_after = Pt(6)
  144. pf.first_line_indent = Pt(0)
  145. if name == 'Heading 1':
  146. pf.page_break_before = True
  147. pf.space_after = Pt(12)
  148. sec = doc.sections[0]
  149. sec.page_width, sec.page_height = Cm(21.0), Cm(29.7)
  150. sec.top_margin = sec.bottom_margin = Cm(2.54)
  151. sec.left_margin = sec.right_margin = Cm(3.17)
  152. def add_field(paragraph, instr: str, placeholder: str = ''):
  153. r = paragraph.add_run()
  154. b = OxmlElement('w:fldChar'); b.set(qn('w:fldCharType'), 'begin')
  155. i = OxmlElement('w:instrText'); i.set(qn('xml:space'), 'preserve'); i.text = instr
  156. s = OxmlElement('w:fldChar'); s.set(qn('w:fldCharType'), 'separate')
  157. t = OxmlElement('w:t'); t.text = placeholder
  158. e = OxmlElement('w:fldChar'); e.set(qn('w:fldCharType'), 'end')
  159. for el in (b, i, s, t, e):
  160. r._r.append(el)
  161. return r
  162. def para(doc, text: str = '', *, size=SZ['body'], ea=SONG, bold=False, align='left',
  163. indent_chars=0, space_before=0, space_after=6, hanging=False):
  164. p = doc.add_paragraph()
  165. pf = p.paragraph_format
  166. pf.line_spacing_rule = WD_LINE_SPACING.MULTIPLE
  167. pf.line_spacing = 1.5
  168. pf.space_before = Pt(space_before)
  169. pf.space_after = Pt(space_after)
  170. pf.first_line_indent = Pt(size * indent_chars)
  171. if hanging:
  172. pf.left_indent = Cm(0.74)
  173. pf.first_line_indent = Cm(-0.0)
  174. pf.alignment = {'left': WD_ALIGN_PARAGRAPH.LEFT, 'center': WD_ALIGN_PARAGRAPH.CENTER,
  175. 'justify': WD_ALIGN_PARAGRAPH.JUSTIFY}[align]
  176. if text:
  177. for chunk, is_bold, mono in split_bold(text):
  178. r = p.add_run(chunk)
  179. set_run(r, size, ea=SONG, ascii_font='Consolas' if mono else WEST, bold=bold or is_bold)
  180. return p
  181. INLINE_RE = re.compile(r'\*\*(.+?)\*\*|`([^`]+)`')
  182. CODE_RE = re.compile(r'`([^`]+)`')
  183. BOLD_RE = re.compile(r'\*\*(.+?)\*\*')
  184. def _bold_parts(s: str):
  185. parts, pos = [], 0
  186. for m in BOLD_RE.finditer(s):
  187. if m.start() > pos:
  188. parts.append((s[pos:m.start()], False, False))
  189. parts.append((m.group(1), True, False))
  190. pos = m.end()
  191. tail = s[pos:]
  192. if '**' in tail:
  193. # 落单的 `**`(例如通配路径 `pitch/**` 没包在行内代码里)—— 交付文档里不能留字面量,
  194. # 去掉并记一条告警,让写手回去补行内代码。
  195. WARN.append('落单的 ** 已剔除: ' + tail.strip()[:60])
  196. tail = tail.replace('**', '')
  197. if tail:
  198. parts.append((tail, False, False))
  199. return parts
  200. def split_bold(text: str):
  201. """行内标记 → [(片段, 是否粗体, 是否等宽)]。
  202. ★单遍分词,支持两种嵌套(2026-09-22 两次实逮后的定稿):
  203. ① 代码里含 `**`(如通配路径 `` `pitch/**` ``)—— 不能当成粗体起始;
  204. ② 粗体里含代码(如 ``**体量以 `data/raw` 为准**``)—— 不能按代码先切、把粗体配对切断。
  205. 先按 `**…**` 找配对,配对内部再递归扫代码;代码段内部一律**字面**取,不再解析粗体。
  206. 落单的 `**` 剔除并记告警(交付文档里不能留字面星号)。
  207. """
  208. out: list[tuple[str, bool, bool]] = []
  209. def emit(t: str, bold: bool, mono: bool):
  210. if t:
  211. out.append((t, bold, mono))
  212. def scan(s: str, bold: bool):
  213. i = 0
  214. while i < len(s):
  215. if s.startswith('**', i):
  216. j = s.find('**', i + 2)
  217. if j > i + 2:
  218. scan(s[i + 2:j], True)
  219. i = j + 2
  220. continue
  221. WARN.append('落单的 ** 已剔除: ' + s[max(0, i - 20):i + 20].strip())
  222. i += 2
  223. continue
  224. if s[i] == '`':
  225. j = s.find('`', i + 1)
  226. if j > i:
  227. emit(s[i + 1:j], bold, True)
  228. i = j + 1
  229. continue
  230. # ★未配对的反引号:当字面字符处理,并**必须前进 i**。
  231. # 2026-09-28 实逮(设计说明渲染满核 30 分钟不出件,faulthandler 栈落在下面那行 emit):
  232. # 原先这里不前进,掉到下面的"找下一个标记"分支 —— 而 `range(i, len(s))` 在 k=i 就命中
  233. # 这个反引号 ⇒ `nxt = i` ⇒ `while` 永不退出。落单标记只该报 WARN,不该把渲染挂死。
  234. WARN.append('落单的 ` 已按字面处理: ' + s[max(0, i - 20):i + 20].strip())
  235. emit('`', bold, False)
  236. i += 1
  237. continue
  238. nxt = len(s)
  239. for k in range(i, len(s)):
  240. if s[k] == '`' or s.startswith('**', k):
  241. nxt = k
  242. break
  243. if nxt <= i: # 兜底:任何情况下都不许原地打转
  244. nxt = i + 1
  245. emit(s[i:nxt], bold, False)
  246. i = nxt
  247. scan(text, False)
  248. return out or [(text, False, False)]
  249. def shade(cell, fill='F2F2F2'):
  250. tcPr = cell._tc.get_or_add_tcPr()
  251. shd = OxmlElement('w:shd')
  252. shd.set(qn('w:val'), 'clear')
  253. shd.set(qn('w:color'), 'auto')
  254. shd.set(qn('w:fill'), fill)
  255. tcPr.append(shd)
  256. def repeat_header(row):
  257. trPr = row._tr.get_or_add_trPr()
  258. th = OxmlElement('w:tblHeader')
  259. th.set(qn('w:val'), 'true')
  260. trPr.append(th)
  261. def add_table(doc, rows: list[list[str]]):
  262. ncol = max(len(r) for r in rows)
  263. t = doc.add_table(rows=0, cols=ncol)
  264. t.style = 'Table Grid'
  265. t.alignment = WD_TABLE_ALIGNMENT.CENTER
  266. t.autofit = True
  267. size = SZ['table_small'] if ncol >= 6 else SZ['table']
  268. for i, r in enumerate(rows):
  269. cells = t.add_row().cells
  270. for j in range(ncol):
  271. txt = (r[j] if j < len(r) else '').strip()
  272. cell = cells[j]
  273. cell.text = ''
  274. p = cell.paragraphs[0]
  275. p.paragraph_format.line_spacing = 1.15
  276. p.paragraph_format.space_after = Pt(0)
  277. p.paragraph_format.first_line_indent = Pt(0)
  278. p.alignment = WD_ALIGN_PARAGRAPH.CENTER if i == 0 else WD_ALIGN_PARAGRAPH.LEFT
  279. for chunk, is_bold, mono in split_bold(txt):
  280. run = p.add_run(chunk.replace('\\|', '|'))
  281. set_run(run, size, ea=HEI if i == 0 else SONG,
  282. ascii_font='Consolas' if mono else WEST, bold=(i == 0) or is_bold)
  283. if i == 0:
  284. shade(cell)
  285. if t.rows:
  286. repeat_header(t.rows[0])
  287. return t
  288. # ── markdown 解析 ──────────────────────────────────────────────────────────────
  289. IMG_RE = re.compile(r'^!\[(?P<cap>[^\]]*)\]\((?P<path>[^)]+)\)\s*$')
  290. CAP_RE = re.compile(r'^\*\*(?P<cap>表\s*[\d\-–.]+.*?)\*\*\s*$')
  291. def parse_md(text: str):
  292. """→ [('h1'|'h2'|'h3'|'h4'|'p'|'ul'|'ol'|'table'|'img'|'quote', payload)]"""
  293. lines = text.splitlines()
  294. blocks, i = [], 0
  295. if lines and lines[0].startswith('# '):
  296. i = 1
  297. while i < len(lines):
  298. ln = lines[i]
  299. s = ln.strip()
  300. if not s or re.match(r'^([-*_])(\s*\1){2,}$', s):
  301. i += 1 # 空行与分隔线(`---` / `***` / `___`)
  302. continue
  303. m = re.match(r'^(#{2,5})\s+(.*)$', s)
  304. if m:
  305. blocks.append(('h%d' % (len(m.group(1)) - 1), m.group(2).strip()))
  306. i += 1
  307. continue
  308. m = IMG_RE.match(s)
  309. if m:
  310. blocks.append(('img', (m.group('cap'), m.group('path'))))
  311. i += 1
  312. continue
  313. if s.startswith('```') or s.startswith('~~~'):
  314. # ★2026-09-28:交付文档里首次出现代码栅栏(P9 的目录树)—— 原先没有栅栏处理,
  315. # 栅栏行会被当普通段落,行内解析器遇到落单的 ` 直接死循环(见 split_bold 的注释)。
  316. fence = s[:3]
  317. i += 1
  318. code = []
  319. while i < len(lines) and not lines[i].strip().startswith(fence):
  320. code.append(lines[i])
  321. i += 1
  322. i += 1 # 跳过收尾栅栏(缺失也能收敛)
  323. blocks.append(('code', code))
  324. continue
  325. if s.startswith('|'):
  326. rows = []
  327. while i < len(lines) and lines[i].strip().startswith('|'):
  328. raw = lines[i].strip().strip('|')
  329. if not re.match(r'^[\s:\-|]+$', raw):
  330. rows.append([c.strip() for c in raw.split('|')])
  331. i += 1
  332. if rows:
  333. cap = None
  334. if blocks and blocks[-1][0] == 'p' and CAP_RE.match(blocks[-1][1]):
  335. cap = blocks.pop()[1]
  336. blocks.append(('table', (cap, rows)))
  337. continue
  338. if s.startswith('- '):
  339. items = []
  340. while i < len(lines) and lines[i].strip().startswith('- '):
  341. items.append(lines[i].strip()[2:].strip())
  342. i += 1
  343. blocks.append(('ul', items))
  344. continue
  345. if re.match(r'^\d+[.)]\s+', s):
  346. items = []
  347. while i < len(lines) and re.match(r'^\d+[.)]\s+', lines[i].strip()):
  348. items.append(re.sub(r'^\d+[.)]\s+', '', lines[i].strip()))
  349. i += 1
  350. blocks.append(('ol', items))
  351. continue
  352. if s.startswith('> '):
  353. q = []
  354. while i < len(lines) and lines[i].strip().startswith('> '):
  355. q.append(lines[i].strip()[2:].strip())
  356. i += 1
  357. blocks.append(('quote', ' '.join(q)))
  358. continue
  359. blocks.append(('p', s))
  360. i += 1
  361. return blocks
  362. def forbidden_hits(text: str) -> dict:
  363. """扫去标识化词表 → {禁用词: 命中次数}(只回报非零项;**大小写不敏感**)。"""
  364. low = text.lower()
  365. return {w: low.count(w.lower()) for w in FORBIDDEN if low.count(w.lower())}
  366. def strict_hits(text: str) -> dict:
  367. """数据要求说明的"不得出现实现细节/现状"扫描 → {模式名: 命中次数}(只回报非零项)。"""
  368. return {name: len(re.findall(rx, text, re.I)) for name, rx in STRICT_PATTERNS
  369. if re.search(rx, text, re.I)}
  370. def count_md(text: str) -> dict:
  371. blocks = parse_md(text)
  372. body = re.sub(r'!\[[^\]]*\]\([^)]*\)', '', text)
  373. cn = len(re.findall(r'[\u4e00-\u9fff]', body))
  374. return dict(
  375. 章=sum(1 for k, _ in blocks if k == 'h1'),
  376. 节=sum(1 for k, _ in blocks if k == 'h2'),
  377. 小节=sum(1 for k, _ in blocks if k == 'h3'),
  378. 表=sum(1 for k, _ in blocks if k == 'table'),
  379. 图=sum(1 for k, _ in blocks if k == 'img'),
  380. 段落=sum(1 for k, _ in blocks if k == 'p'),
  381. 汉字=cn)
  382. # ── 渲染 ───────────────────────────────────────────────────────────────────────
  383. def cover(doc, title: str, subtitle: str):
  384. for _ in range(4):
  385. para(doc, '', space_after=0)
  386. para(doc, f'{title}', size=SZ['cover'], ea=HEI, bold=True, align='center', space_after=12)
  387. para(doc, subtitle, size=14, ea=HEI, align='center', space_after=6)
  388. for _ in range(3):
  389. para(doc, '', space_after=0)
  390. para(doc, f'版本 {VERSION}(与系统版本一致)', size=14, align='center', space_after=6)
  391. para(doc, f'编制:观澜项目组  日期:2026-09-22', size=12, align='center', space_after=6)
  392. # ★落款不写生成器文件名与源件路径:那会把实现细节带进"对外、不体现文件名/落盘位置"的文档里
  393. # (2026-09-22 实逮: `--check` 扫源件是 clean, 但 `--verify` 扫渲染结果命中 1 处 snake_case + 2 处路径)。
  394. para(doc, '本文件由统一的交付文档生成器排版(样式表固定,可重跑、可 diff;正文源件随交付包提供)',
  395. size=9, align='center', space_after=0)
  396. doc.add_page_break()
  397. def toc(doc):
  398. para(doc, '目录', size=SZ['h1'], ea=HEI, bold=True, align='center', space_after=12)
  399. p = doc.add_paragraph()
  400. p.paragraph_format.line_spacing = 1.5
  401. add_field(p, r'TOC \o "1-3" \h \z \u',
  402. '(目录域:在 Word 中按 Ctrl+A 后 F9,或右键“更新域”即可生成带页码的目录)')
  403. doc.add_page_break()
  404. def render_blocks(doc, blocks, fig_root: pathlib.Path):
  405. """渲染块序列。表号一律**按章编号**「表 <章>-<序>」(附录按字母 A/B/C)——
  406. 与三份稿正文里的引用口径一致(实逮: 原先按出现顺序编成「表 1…表 54」, 正文写「见表 9-2」就指不到)。
  407. """
  408. tno = 0
  409. chap_label = ''
  410. per = 0
  411. caps: list[str] = []
  412. for kind, payload in blocks:
  413. if kind == 'h1':
  414. head = payload.strip()
  415. m = re.match(r'^(\d+)', head)
  416. if m:
  417. chap_label, per = m.group(1), 0
  418. else:
  419. m2 = re.match(r'^附录\s*([A-Za-z])', head)
  420. chap_label, per = (m2.group(1), 0) if m2 else (chap_label, 0)
  421. if kind.startswith('h'):
  422. lvl = int(kind[1])
  423. style = {1: 'Heading 1', 2: 'Heading 2', 3: 'Heading 3', 4: 'Heading 4'}[lvl]
  424. p = doc.add_paragraph(style=style)
  425. size = {1: SZ['h1'], 2: SZ['h2'], 3: SZ['h3'], 4: SZ['h3']}[lvl]
  426. for chunk, _b, _m in split_bold(payload):
  427. set_run(p.add_run(chunk), size, ea=HEI, bold=True)
  428. continue
  429. if kind == 'p':
  430. para(doc, payload, align='justify', indent_chars=2)
  431. continue
  432. if kind in ('ul', 'ol'):
  433. for n, item in enumerate(payload, 1):
  434. mark = '· ' if kind == 'ul' else f'({n})'
  435. para(doc, mark + item, align='justify', hanging=True, space_after=3)
  436. continue
  437. if kind == 'quote':
  438. p = para(doc, payload, align='justify', space_before=6, space_after=6)
  439. p.paragraph_format.left_indent = Cm(0.74)
  440. continue
  441. if kind == 'code':
  442. # 代码/目录树块:等宽、小一号、逐行不折行缩进(交付文档里用于"目录结构""命令示例")
  443. for ln in payload:
  444. p = doc.add_paragraph()
  445. p.paragraph_format.line_spacing = 1.0
  446. p.paragraph_format.space_before = Pt(0)
  447. p.paragraph_format.space_after = Pt(0)
  448. p.paragraph_format.left_indent = Cm(0.4)
  449. set_run(p.add_run(ln if ln.strip() else ' '), SZ['table_small'], ascii_font='Consolas')
  450. para(doc, '', space_after=4)
  451. continue
  452. if kind == 'table':
  453. cap, rows = payload
  454. tno += 1
  455. per += 1
  456. label = f'表 {chap_label}-{per}' if chap_label else f'表 {tno}'
  457. title = re.sub(r'^表\s*[\dA-Za-z]+\s*[-–—]\s*\d+\s*', '', cap or '').strip()
  458. if cap:
  459. author_no = re.match(r'^表\s*([\dA-Za-z]+)\s*[-–—]\s*(\d+)', cap)
  460. if author_no and f'{author_no.group(1)}-{author_no.group(2)}' != f'{chap_label}-{per}':
  461. WARN.append('表号与作者自编不一致: 原「%s」→ 本器按章编号「%s」'
  462. % (cap[:24], label))
  463. caps.append(label)
  464. para(doc, (label + (' ' + title if title else '')), size=SZ['caption'], ea=HEI, bold=True,
  465. align='center', space_before=8, space_after=4)
  466. add_table(doc, rows)
  467. para(doc, '', space_after=2)
  468. continue
  469. if kind == 'img':
  470. cap, rel = payload
  471. # md 里的图路径是相对 `docs/` 的(约定 `figures/xxx.png`);容错:只给文件名也认。
  472. cand = fig_root / rel if (fig_root / rel).exists() else fig_root / 'figures' / pathlib.Path(rel).name
  473. f = cand.resolve()
  474. if f.exists():
  475. p = doc.add_paragraph()
  476. p.alignment = WD_ALIGN_PARAGRAPH.CENTER
  477. p.paragraph_format.space_before = Pt(8)
  478. p.paragraph_format.space_after = Pt(4)
  479. p.paragraph_format.first_line_indent = Pt(0)
  480. p.add_run().add_picture(str(f), width=Cm(15.0) if f.stat().st_size > 90_000 else Cm(13.0))
  481. else:
  482. WARN.append(f'图缺件: {rel}')
  483. para(doc, f'(图缺失:{rel})', size=SZ['caption'], align='center', space_after=4)
  484. para(doc, cap or rel, size=SZ['caption'], ea=HEI, bold=True, align='center', space_after=8)
  485. continue
  486. doc._tab_labels = caps # 供调用方自检(本器 --verify 直接从 docx 回读,不依赖它)
  487. return tno
  488. def header_footer(doc, title: str):
  489. sec = doc.sections[0]
  490. hp = sec.header.paragraphs[0]
  491. hp.alignment = WD_ALIGN_PARAGRAPH.CENTER
  492. set_run(hp.add_run(f'{title} · 观澜 v2 风电场智能分析系统 · 版本 {VERSION}'), SZ['header'])
  493. fp = sec.footer.paragraphs[0]
  494. fp.alignment = WD_ALIGN_PARAGRAPH.CENTER
  495. set_run(fp.add_run('第 '), SZ['header'])
  496. add_field(fp, 'PAGE', '1')
  497. set_run(fp.add_run(' 页 / 共 '), SZ['header'])
  498. add_field(fp, 'NUMPAGES', '1')
  499. set_run(fp.add_run(' 页'), SZ['header'])
  500. def appendix_spec(doc):
  501. para(doc, '附:排版规范(字体字号分类)', size=SZ['h1'], ea=HEI, bold=True, align='left',
  502. space_before=12, space_after=8)
  503. para(doc, '本文件与同批交付的另外两份文档由同一个渲染器生成,字体字号分类完全一致;'
  504. '中文用黑体(标题)/宋体(正文),西文与数字用 Times New Roman。', align='justify',
  505. indent_chars=2)
  506. add_table(doc, [['元素', '字体', '字号', '对齐 / 缩进']] + [list(r) for r in SPEC_ROWS])
  507. def build(key: str) -> int:
  508. spec = DOCS[key]
  509. src = ROOT / spec['src'].format(v=VERSION)
  510. out = ROOT / spec['out'].format(v=VERSION)
  511. if not src.exists():
  512. WARN.append(f'源件缺件: {src.relative_to(ROOT).as_posix()}')
  513. print(f'✗ {key}: 源件不在位 {src}')
  514. return 5
  515. blocks = parse_md(src.read_text(encoding='utf-8'))
  516. doc = Document()
  517. style_setup(doc)
  518. cover(doc, spec['title'], spec['subtitle'])
  519. toc(doc)
  520. ntab = render_blocks(doc, blocks, ROOT / 'docs')
  521. appendix_spec(doc)
  522. header_footer(doc, spec['title'])
  523. out.parent.mkdir(parents=True, exist_ok=True)
  524. doc.save(out)
  525. st = count_md(src.read_text(encoding='utf-8'))
  526. print('%s → %s(%d 章 · %d 节 · %d 表 · %d 图 · 约 %d 汉字 · %.0f KB)'
  527. % (key, out.relative_to(ROOT).as_posix(), st['章'], st['节'], max(ntab, st['表']), st['图'],
  528. st['汉字'], out.stat().st_size / 1024))
  529. return 0
  530. def verify_docx(key: str) -> int:
  531. """回读渲染结果自检(交付前必跑):计数 + 残留 markdown 标记 + 域 + 字体。"""
  532. spec = DOCS[key]
  533. out = ROOT / spec['out'].format(v=VERSION)
  534. if not out.is_file():
  535. print('✗ %-4s 不在位 %s' % (key, out.relative_to(ROOT).as_posix()))
  536. return 5
  537. doc = Document(str(out))
  538. texts = [p.text for p in doc.paragraphs]
  539. cells = [c.text for t in doc.tables for r in t.rows for c in r.cells]
  540. anti = chr(96)
  541. # ★行内代码里的字面量不算残留(实逮: `` `pitch/**` `` 是 glob 通配路径,Word 里就该原样显示)。
  542. mono, plain = [], []
  543. for p in doc.paragraphs:
  544. for r in p.runs:
  545. (mono if (r.font.name or '').startswith('Consolas') else plain).append(r.text)
  546. for t in doc.tables:
  547. for row in t.rows:
  548. for c in row.cells:
  549. for p in c.paragraphs:
  550. for r in p.runs:
  551. (mono if (r.font.name or '').startswith('Consolas') else plain).append(r.text)
  552. allt = '\n'.join(texts + cells)
  553. body = '\n'.join(plain)
  554. heads = [p.text for p in doc.paragraphs if p.style.name.startswith('Heading')]
  555. residual = dict(反引号=body.count(anti), 星号=body.count('**'), 管道=body.count(' | '),
  556. 代码内字面星号='\n'.join(mono).count('**'))
  557. foot = doc.sections[0].footer.paragraphs[0]._p.xml
  558. head = doc.sections[0].header.paragraphs[0].text
  559. ok_field = ('TOC' in doc.element.xml) and ('PAGE' in foot) and ('NUMPAGES' in foot)
  560. # 只把"正文里残留的 markdown 标记"判为不合格;代码内的字面星号是内容本身(glob 通配),只报数。
  561. bad = [k for k, v in residual.items() if v and k != '代码内字面星号']
  562. # ★版本表覆盖检查(2026-09-22 实逮): 升版本时若只改版本串而忘插表行, 表尾会停在旧版本,
  563. # 而正文"共 N 条/十六个版本/中七条·小九条"是手写的 ⇒ 用 HISTORY 机器核对, 缺一行就 FAIL。
  564. if key == 'req':
  565. from src import version as _V
  566. want_v = {h['version'] for h in _V.HISTORY}
  567. got_v = set()
  568. for _tb in doc.tables:
  569. for _row in _tb.rows:
  570. _c = _row.cells[0].text.strip()
  571. if re.fullmatch(r'\d+\.\d+\.\d+', _c):
  572. got_v.add(_c)
  573. _miss_v = sorted(want_v - got_v)
  574. if _miss_v:
  575. bad.append('版本表缺行 %s' % _miss_v)
  576. leak = forbidden_hits(allt + '\n' + head) # ★对外件不得出现样本场标识
  577. strict = strict_hits(allt) if spec.get('strict_terms') else {} # ★数据要求说明: 去实现细节/去现状
  578. # ★表号断链检查(2026-09-22 实逮): 正文写「见表 9-2」,文档里就必须真有「表 9-2」。
  579. cap_labels = set()
  580. for p in doc.paragraphs:
  581. mm = re.match(r'^表\s*([\dA-Za-z]+)\s*[-–—]\s*(\d+)', p.text.strip())
  582. if mm:
  583. cap_labels.add(f'{mm.group(1)}-{mm.group(2)}')
  584. def _refs(text: str) -> set:
  585. """正文里的「表 X-Y」引用。
  586. ★两处实逮的误报都要挡:① `年-月`('报表 2026-09');② **跨单元格拼接** —— 把表格单元用换行
  587. 连起来后, "…如实空表" + "FR-18 至 FR-21" 会拼出 '表\\nFR-18',被当成表号。故分隔符只许空格/制表符。
  588. """
  589. out = set()
  590. for a, b in re.findall(r'表[ \t]*([\dA-Za-z]+)[ \t]*[-–—][ \t]*(\d+)', text):
  591. if re.fullmatch(r'(19|20)\d{2}', a):
  592. continue
  593. out.add(f'{a}-{b}')
  594. return out
  595. refs = _refs('\n'.join(
  596. [p.text for p in doc.paragraphs
  597. if not re.match(r'^(表|图)\s*[\dA-Za-z]+\s*[-–—]\s*\d+', p.text.strip())] + cells))
  598. broken = sorted(refs - cap_labels)
  599. noref = sorted(cap_labels - refs)
  600. print('%-4s %-38s 段落 %3d · 表 %2d · 图 %2d · 标题 %2d · 页眉 %r'
  601. % (key, out.name, len(doc.paragraphs), len(doc.tables), len(doc.inline_shapes),
  602. len(heads), head[:28]))
  603. print(' 域(目录/页码): %s · 残留标记: %s · 去标识化: %s%s · 表号: %s'
  604. % ('OK' if ok_field else '缺', bad or '无',
  605. 'clean' if not leak else '泄漏 %s' % leak,
  606. '' if not spec.get('strict_terms') else
  607. ' · 去实现细节: %s' % ('clean' if not strict else '命中 %s' % strict),
  608. '断链 %s' % broken if broken else '引用闭合(表 %d 张,%d 张无引用)' % (len(cap_labels), len(noref))))
  609. return 0 if ok_field and not bad and not leak and not broken and not strict else 5
  610. def main() -> int:
  611. ap = argparse.ArgumentParser()
  612. ap.add_argument('--only', choices=sorted(DOCS))
  613. ap.add_argument('--check', action='store_true', help='只体检源件与图,不渲染')
  614. ap.add_argument('--verify', action='store_true', help='渲染后回读自检(计数/残留标记/域)')
  615. ap.add_argument('--spec', action='store_true', help='打印排版规范表')
  616. a = ap.parse_args()
  617. if a.spec:
  618. for row in SPEC_ROWS:
  619. print(' | '.join(row))
  620. return 0
  621. rc = 0
  622. if a.verify:
  623. for k in ([a.only] if a.only else sorted(DOCS)):
  624. rc = max(rc, verify_docx(k))
  625. return rc
  626. if a.check:
  627. for k, spec in DOCS.items():
  628. src = ROOT / spec['src'].format(v=VERSION)
  629. if not src.exists():
  630. print('✗ %-4s 源件不在位 %s' % (k, src.relative_to(ROOT).as_posix()))
  631. rc = 5
  632. continue
  633. st = count_md(src.read_text(encoding='utf-8'))
  634. _src_txt = src.read_text(encoding='utf-8')
  635. miss, leak = [], forbidden_hits(_src_txt)
  636. strict = strict_hits(_src_txt) if spec.get('strict_terms') else {}
  637. for _cap, rel in re.findall(r'!\[([^\]]*)\]\(([^)]+)\)', _src_txt):
  638. f = ROOT / 'docs' / rel.split('figures/')[-1]
  639. if not (ROOT / 'docs' / 'figures' / rel.split('figures/')[-1]).exists():
  640. miss.append(rel)
  641. print('%-4s 章 %2d · 节 %2d · 小节 %2d · 表 %2d · 图 %2d · 约 %5d 汉字 · 去标识化 %s%s%s'
  642. % (k, st['章'], st['节'], st['小节'], st['表'], st['图'], st['汉字'],
  643. 'clean' if not leak else '泄漏 %s' % leak,
  644. '' if not spec.get('strict_terms') else
  645. ' · 去实现细节 %s' % ('clean' if not strict else '命中 %s' % strict),
  646. '' if not miss else ' ✗ 缺图 %d: %s' % (len(miss), ', '.join(miss))))
  647. if miss or leak or strict:
  648. rc = 5
  649. return rc
  650. for k in ([a.only] if a.only else sorted(DOCS)):
  651. rc = max(rc, build(k))
  652. if WARN:
  653. print('\n告警:')
  654. for w in sorted(set(WARN)):
  655. print(' -', w)
  656. return rc
  657. if __name__ == '__main__':
  658. raise SystemExit(main())