|
|
@@ -104,15 +104,6 @@ FAMILIES: list[dict] = [
|
|
|
kind='shipped',
|
|
|
func='本体层 · 领域数据件(油样台账/备件价格/场景修复)', algo='(无生成端, 领域正本出件)',
|
|
|
gen=None, input=None, pred=(), why='领域正本随包发来, 不由 data/raw 推导'),
|
|
|
- dict(id='windscada_delivery_notes',
|
|
|
- glob=['windscada/mblub_alarm_cross.json', 'windscada/release-manifest.json',
|
|
|
- 'windscada/review_presented.json', 'windscada/给振动线_温度轴回复_20260826.json',
|
|
|
- 'windscada/给振动线_温度轴回复_20260826.md', 'windscada/送审_系统结论与判定_20260828.md'],
|
|
|
- kind='not-product',
|
|
|
- func='**非产物**:交证/评审往来件(送审结论、给振动线的回复、发布清单快照)',
|
|
|
- algo='(沟通与评审留痕, 不是产物)', gen=None, input=None, pred=(),
|
|
|
- why='这些是"人对人的交证件", 内容由分析结论抄写而成, 不随 data/raw 变; 放在产物仓里会让'
|
|
|
- '"产物"这个词失去意义(它们既无生成端, 也无重算入口)'),
|
|
|
|
|
|
# ── 以下族: 全库 0 处写入方 ⇒ 反向呼应**在原理上不成立**(如实记账, 不假装成立)────
|
|
|
dict(id='vib_handoff_and_scans', glob='m5_cms_tcm/*.{json,parquet,csv,md,txt}',
|
|
|
@@ -149,13 +140,30 @@ FAMILIES: list[dict] = [
|
|
|
func='事实契约与对外派生(门户结论段/取数)', algo='scripts/guanlan_facts_contract.py(读者在包内, '
|
|
|
'输入是 ontology+m5+sop 三处**产物**)', gen=None, input=None, pred=(),
|
|
|
why='派生链的中间层:它的"输入"本身是产物而非 data/raw ⇒ 与 raw 之间隔着不止一跳'),
|
|
|
+ dict(id='windscada_release_manifest', glob='windscada/release-manifest.json', kind='source-derived',
|
|
|
+ func='发布清单快照(`/healthz` 与版本信息读它)',
|
|
|
+ algo='scripts/guanlan_baseline_manifest.py(从 src/windscada/ui 源重建 zh 页 → 摘要 '
|
|
|
+ '{css_tokens, nav, dom, sha256, source blobs})',
|
|
|
+ gen='scripts/guanlan_baseline_manifest.py', input=None, pred=(),
|
|
|
+ why='★ 2026-09-17 修正: 它一度被我按"交证件"错分成非产物; 实际**包内有生成端**'
|
|
|
+ '(guanlan_baseline_manifest.py 就写这个路径), 且 guanlan_gateway.py 运行期读它 ⇒ 是产物'),
|
|
|
+ dict(id='human_deliverables',
|
|
|
+ glob=['windscada/mblub_alarm_cross.json', 'windscada/review_presented.json',
|
|
|
+ 'windscada/给振动线_温度轴回复_20260826.json', 'windscada/给振动线_温度轴回复_20260826.md',
|
|
|
+ 'windscada/送审_系统结论与判定_20260828.md'],
|
|
|
+ kind='human',
|
|
|
+ func='**人工正本/交证件**:与振动线的跨线通知、送审结论、发布评审快照',
|
|
|
+ algo='(人写的, 不是算出来的)', gen=None, input=None, pred=(),
|
|
|
+ why='这几件是"人对人的交证件/回执", 内容由分析结论抄写与人判断坐实而成; '
|
|
|
+ 'mblub_alarm_cross.json 还被 src/ontology/populate.py 当作**证据来源**引用 ⇒ '
|
|
|
+ '必须随包留在原位, 但**不参与呼应校验**(它本来就没有生成端, 也不该有)'),
|
|
|
dict(id='scratch_in_products', glob=['**/_*.js', '**/_*audit*.md', '**/*.out', '**/*.err',
|
|
|
'**/*_test_*.parquet', '**/*_test_*.json'],
|
|
|
kind='not-product',
|
|
|
func='**非产物**:自审脚本/记录与模型测试输出(躺在产物仓里)', algo='(工具脚本与运行痕迹, 不是产物)',
|
|
|
gen=None, input=None, pred=(),
|
|
|
- why='清单里既有审计脚本(`windscada/_audit*.js`)与自审记录(`_*audit*.md`), 也有模型试跑输出'
|
|
|
- '(`ontology/*_test_*.out/.err`);它们随包发过来、不参与任何取数, 属"产物仓里的杂物"'),
|
|
|
+ why='2026-09-17: 17 件已移出到 reference/非产物留痕_20260917/ 并留了 _来源说明.json ⇒ '
|
|
|
+ '本族现在应为空; 一旦再出现(新脚本往产物仓里写 .out/_audit.js 之类), 本器当场报出来'),
|
|
|
]
|
|
|
|
|
|
|
|
|
@@ -352,6 +360,67 @@ def _input_turbines(station: pathlib.Path, sub: str):
|
|
|
return out or None
|
|
|
|
|
|
|
|
|
+def classify_human(farm: str, rels: list[str], cap: int = 600) -> set[str]:
|
|
|
+ """逐件判"是不是人工件"(用户令 2026-09-17 #3:人工件不再按自动产物参与呼应校验)。
|
|
|
+
|
|
|
+ 判据:**非测量形态**的件(.md/.json/.csv/.txt)里出现人工判断字段(按字段名匹配)。
|
|
|
+ 测量形态件(.parquet/.csv 数值、.npz)一律不算人工件 —— 它们是数据的函数,属"可反推"那一路。
|
|
|
+ """
|
|
|
+ human: set[str] = set()
|
|
|
+ for rel in rels[:cap]:
|
|
|
+ suf = pathlib.Path(rel).suffix.lower()
|
|
|
+ if suf in ('.md', '.txt'):
|
|
|
+ pat = JUDGE_PAT
|
|
|
+ elif suf in ('.json', '.csv'):
|
|
|
+ pat = JUDGE_KEY_PAT
|
|
|
+ else:
|
|
|
+ continue
|
|
|
+ p = P.out_root(farm) / rel
|
|
|
+ try:
|
|
|
+ txt = p.read_text(encoding='utf-8', errors='replace')[:120000]
|
|
|
+ except Exception:
|
|
|
+ continue
|
|
|
+ if pat.search(txt):
|
|
|
+ human.add(rel)
|
|
|
+ return human
|
|
|
+
|
|
|
+
|
|
|
+def write_human_manifest(farm: str | None = None) -> int:
|
|
|
+ """把人工件逐件落成 `outputs/<场>/_human_artifacts.json`(机器台账,供自检与审计引用)。"""
|
|
|
+ farm = farm or P.farm()
|
|
|
+ global LEDGER
|
|
|
+ LEDGER = load_ledger(farm)
|
|
|
+ entries: dict[str, dict] = {}
|
|
|
+ for fam in FAMILIES:
|
|
|
+ if fam['kind'] not in ('shipped', 'human', 'not-product'):
|
|
|
+ continue
|
|
|
+ rels = [r for r in _files_of(fam['glob']) if r not in entries]
|
|
|
+ if not rels:
|
|
|
+ continue
|
|
|
+ if fam['kind'] == 'human':
|
|
|
+ for r in rels:
|
|
|
+ entries[r] = dict(kind='human', family=fam['id'], evidence='族内显式登记为人工正本/交证件')
|
|
|
+ continue
|
|
|
+ for r in classify_human(farm, rels):
|
|
|
+ if r not in entries:
|
|
|
+ entries[r] = dict(kind='human', family=fam['id'],
|
|
|
+ evidence='件内含人工判断字段(按字段名匹配),不是输入的纯函数')
|
|
|
+ out = P.out_root(farm) / '_human_artifacts.json'
|
|
|
+ out.write_text(json.dumps(dict(
|
|
|
+ at=__import__('time').strftime('%Y-%m-%d %H:%M:%S'),
|
|
|
+ note='人工件台账(用户令 2026-09-17 #3):这些件由人写成/坐实,不是任何输入的纯函数 ⇒ '
|
|
|
+ '不参与"输出↔输入呼应"校验;随包留在原位,重算不会也不该重写它们。'
|
|
|
+ '本文件由 scripts/products_reverse_audit.py --write-human-manifest 生成。',
|
|
|
+ count=len(entries), files=entries), ensure_ascii=False, indent=1), encoding='utf-8')
|
|
|
+ by_fam: dict[str, int] = {}
|
|
|
+ for v in entries.values():
|
|
|
+ by_fam[v['family']] = by_fam.get(v['family'], 0) + 1
|
|
|
+ print(f'人工件台账: {len(entries)} 件 → {P.rel(out)}')
|
|
|
+ for k, v in sorted(by_fam.items(), key=lambda kv: -kv[1]):
|
|
|
+ print(f' {k:24s} {v:4d} 件')
|
|
|
+ return 0
|
|
|
+
|
|
|
+
|
|
|
def audit(farm: str | None = None, verbose: bool = True):
|
|
|
farm = farm or P.farm()
|
|
|
global LEDGER
|
|
|
@@ -383,18 +452,35 @@ def audit(farm: str | None = None, verbose: bool = True):
|
|
|
for r in rels:
|
|
|
verdict[r] = dict(fam=fam['id'], verdict='✗',
|
|
|
why='非产物(过程留痕/工具脚本躺在产物仓里): ' + fam.get('why', ''))
|
|
|
+ elif fam['kind'] == 'human':
|
|
|
+ # ★ 人工正本/交证件 (用户令 2026-09-17 #3): 由人写、被运行期当证据引用,
|
|
|
+ # **不参与呼应校验**——它本来就没有、也不该有生成端; 单独计数, 不混进"无生成端"的失败堆里。
|
|
|
+ row['verdict'] = '◆ 人工件'
|
|
|
+ row['note'] = fam.get('why', '')
|
|
|
+ row['rev'], row['rev_why'] = '人工件(正本/交证)', '人写的, 不参与呼应校验; 随包留在原位'
|
|
|
+ for r in rels:
|
|
|
+ verdict[r] = dict(fam=fam['id'], verdict='◆',
|
|
|
+ why='人工件(人工正本/交证件): ' + fam.get('why', ''))
|
|
|
elif fam['kind'] == 'shipped':
|
|
|
jp, js, pg = judgement_evidence(farm, rels)
|
|
|
gen_file = ROOT / fam['gen'] if fam.get('gen') else None
|
|
|
gen_ok0 = bool(gen_file and gen_file.is_file())
|
|
|
- fv, fw = feasibility(fam, rels, gen_ok0, False, jp, js, pg)
|
|
|
- row['verdict'] = '✗ 无生成端'
|
|
|
+ # ★ 逐件分人工件 (用户令 2026-09-17 #3): 含人工判断字段的件**不按自动产物参与呼应校验**,
|
|
|
+ # 单独计 ◆; 余下才是真正的"无生成端"。
|
|
|
+ human = classify_human(farm, rels)
|
|
|
+ rest = [r for r in rels if r not in human]
|
|
|
+ fv, fw = feasibility(fam, rest or rels, gen_ok0, False, jp, js, pg)
|
|
|
+ row['verdict'] = '✗ 无生成端' if not human else f'✗ 无生成端 + ◆ 人工件 {len(human)}'
|
|
|
+ row['n'] = len(rest)
|
|
|
row['note'] = fam.get('why', '')
|
|
|
row['rev'], row['rev_why'] = fv, fw
|
|
|
if jp:
|
|
|
row['rev_why'] += f';人工判断字段抽样命中 {jp}%' + (f'(如 {js[0]})' if js else '')
|
|
|
- for r in rels:
|
|
|
+ for r in rest:
|
|
|
verdict[r] = dict(fam=fam['id'], verdict='✗', why='无生成端(全库 0 处写入方): ' + fam['why'])
|
|
|
+ for r in human:
|
|
|
+ verdict[r] = dict(fam=fam['id'], verdict='◆',
|
|
|
+ why='人工件(含人工判断字段): 不参与呼应校验, 随包留在原位')
|
|
|
else:
|
|
|
# ① 生成端在位
|
|
|
gen = ROOT / fam['gen'] if fam.get('gen') else None
|
|
|
@@ -412,6 +498,11 @@ def audit(farm: str | None = None, verbose: bool = True):
|
|
|
# 跨度口径 —— 技术资料是文档而不是时序数据, 拿跨度比毫无意义)
|
|
|
i_n = sum(1 for p in (home / fam['input']).rglob('*') if p.is_file())
|
|
|
i_gran = '年'
|
|
|
+ elif fam['kind'] == 'source-derived':
|
|
|
+ # ★ 输入是**包内源码**而不是 data/raw(如发布清单快照: 从 src/windscada/ui 重建页再摘要)。
|
|
|
+ # 生成端在位 = 输入在位, 这类产物的"呼应"是对源码的, 不是对原始件的。
|
|
|
+ i_n = 1
|
|
|
+ i_s = i_e = '包内源码'
|
|
|
in_ok = i_n > 0
|
|
|
passed, notes = [], []
|
|
|
if fam['pred'] and in_ok:
|
|
|
@@ -506,7 +597,7 @@ def audit(farm: str | None = None, verbose: bool = True):
|
|
|
print(f' 判据失败 {len(fails)} 条' + (':' + ';'.join(fails[:3]) if fails else ''))
|
|
|
ok = counts.get('✓', 0)
|
|
|
print(f' 结论: 输出↔输入呼应**成立** {ok} 件 / **不成立(无生成端)** {counts.get("✗", 0)} 件 / '
|
|
|
- f'无法验证 {counts.get("~", 0)} 件 / 未归类 {counts.get("?", 0)} 件')
|
|
|
+ f'**人工件** {counts.get("◆", 0)} 件 / 无法验证 {counts.get("~", 0)} 件 / 未归类 {counts.get("?", 0)} 件')
|
|
|
# ── 逆向工程可行性 (用户令 #2): 没有呼应关系的那些, 能不能推出来 ──
|
|
|
import collections as _c
|
|
|
by_rev = _c.Counter()
|
|
|
@@ -553,6 +644,9 @@ def doc_block(farm: str) -> str:
|
|
|
for r in fam_rows:
|
|
|
by_rev[r.get('rev', '?')] += r['n']
|
|
|
L += ['', '**按件数汇总**:' + ' · '.join(f'{k} = {v} 件' for k, v in by_rev.most_common()), '',
|
|
|
+ '**人工件**:含人工判断字段的件已逐件登记在 outputs/<场>/_human_artifacts.json(用户令 2026-09-17 #3),'
|
|
|
+ '它们**不参与**呼应校验(人写成/坐实的东西不是任何输入的纯函数);生成方式:'
|
|
|
+ 'python scripts/products_reverse_audit.py --write-human-manifest。', '',
|
|
|
'> 结论口径:`可逆(直接)` = 包内已有生成端,接上即可;`可逆(需反推口径)` = 内容是测量数据的函数,'
|
|
|
'照 `temp_monthly` 的办法反推 + 逐值对拍;`不可逆(人工判断)` = 逆向工程推不出来,'
|
|
|
'要么由研发补生成端、要么承认它是人工件(不该按产物管)。', DOC_END]
|
|
|
@@ -578,7 +672,11 @@ def main() -> int:
|
|
|
ap.add_argument('--farm', default=None)
|
|
|
ap.add_argument('--check', action='store_true', help='只出结论 (有未归类/判据失败 → rc=5)')
|
|
|
ap.add_argument('--write-doc', action='store_true', help='把族表写进 docs/系统设计说明.md §13.6')
|
|
|
+ ap.add_argument('--write-human-manifest', action='store_true',
|
|
|
+ help='生成人工件台账 outputs/<场>/_human_artifacts.json (用户令 2026-09-17 #3)')
|
|
|
a = ap.parse_args()
|
|
|
+ if a.write_human_manifest:
|
|
|
+ return write_human_manifest(a.farm)
|
|
|
if a.write_doc:
|
|
|
return write_doc(a.farm or P.farm())
|
|
|
_rows, _v, counts, fails, uncl, rc = audit(a.farm, verbose=True)
|