|
@@ -79,6 +79,33 @@ TECH_NOTE = '厂商技术资料(故障处理手册/维护 WI/图纸/对译表
|
|
|
# raw 根下**随包发的说明件**(不是数据): 别把它们报成"未归类"。口径: 只有名字在白名单里且不参与摄入。
|
|
# raw 根下**随包发的说明件**(不是数据): 别把它们报成"未归类"。口径: 只有名字在白名单里且不参与摄入。
|
|
|
IGNORE_FILES = {'README_把原始数据放这里.txt'}
|
|
IGNORE_FILES = {'README_把原始数据放这里.txt'}
|
|
|
|
|
|
|
|
|
|
+# ── 消费者取数口径(用户令 2026-09-20 步骤 C:让"放了没人读"当场可见)────────────────────────
|
|
|
|
|
+# ★为什么必须有它:族表原来只按**扩展名白名单**计件 —— 往 `scada_10min/` 里放 `WTG01-B2.csv`(856 列、
|
|
|
|
|
+# 2026-08 数据、与 `<台号>.csv` 命名不同)时,它命中 `*.csv` ⇒ 既算"进了摄入",又不会出现在任何
|
|
|
|
|
+# "没人读"的提示里;而取数层只认 `<台号>.csv` ⇒ **这批数据从来没被读过**,重算后时间窗自然不变。
|
|
|
|
|
+# 本表按**消费者真实取数口径**再判一次,并把结果分成两级:
|
|
|
|
|
+# · 缺口级 (gate):看着是数据(扩展名在 data_exts 里) 但没有任何消费者会读 ⇒ **计入变化(rc=4) + 大声报**
|
|
|
|
|
+# · 信息级 (info):附件/压缩包/图片/厂商软件这类,本来就不该被摄入 ⇒ 只列信息,不计变化
|
|
|
|
|
+CONSUME: dict[str, dict] = {
|
|
|
|
|
+ '故障报警': dict(consume=('*.xls',), data_exts=('.xls', '.xlsx', '.csv'),
|
|
|
|
|
+ how='摄入器读 *.xls 的 XML 事件导出(TimeOn/Alarmcode)'),
|
|
|
|
|
+ '风机故障记录': dict(consume=('*.xls', '*.xlsx'), data_exts=('.xls', '.xlsx', '.csv'),
|
|
|
|
|
+ how='台账读取 *.xls*(rar/jpg 是附件,不摄入)'),
|
|
|
|
|
+ '油样报告': dict(consume=('*.pdf',), data_exts=('.pdf',), how='油样摄取读 *.pdf'),
|
|
|
|
|
+ 'scada_10min': dict(consume=('WTG??.csv',), data_exts=('.csv', '.mdb', '.zip'),
|
|
|
|
|
+ how='取数层**按台号精确取 `<台号>.csv`**(WTG01.csv);别的命名不会被读'),
|
|
|
|
|
+ 'scada_1min': dict(consume=('???.csv',), data_exts=('.csv', '.mdb', '.zip'),
|
|
|
|
|
+ how='取数层按内部台号取 `<台号>.csv`(01E.csv)'),
|
|
|
|
|
+ 'scada_mdb': dict(consume=('*-10min.mdb', '*-1min.mdb', '*.zip'), data_exts=('.mdb', '.zip'),
|
|
|
|
|
+ how='取数层读取按月的库(2026-08-10min.mdb);CSV 缺失时才回落'),
|
|
|
|
|
+ 'windcms': dict(consume=('*_decode.json', '*.pdf', '*.docx', '*.md', '*.json'),
|
|
|
|
|
+ data_exts=('.json', '.zip', '.csv', '.mdb'),
|
|
|
|
|
+ how='振动摄入读 *_decode.json;厂家报告 PDF 由报告转录侧读'),
|
|
|
|
|
+ 'm5_cms_tcm': dict(consume=('*.json', '*.docx', '*.pdf', '*.md'), data_exts=('.json', '.zip'),
|
|
|
|
|
+ how='现场正本 handoff/component_history 与厂家报告'),
|
|
|
|
|
+ TECH_DIR: dict(consume=('*',), data_exts=(), how='本体读整棵树(白名单=全部)'),
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
|
|
|
def _digest(entries: list[tuple[str, int, int]], deep: bool, base: pathlib.Path, files: list[pathlib.Path],
|
|
def _digest(entries: list[tuple[str, int, int]], deep: bool, base: pathlib.Path, files: list[pathlib.Path],
|
|
|
deep_max: int, dirs: list[str] | None = None) -> str:
|
|
deep_max: int, dirs: list[str] | None = None) -> str:
|
|
@@ -138,8 +165,23 @@ def fingerprint(d: pathlib.Path, patterns: tuple[str, ...], deep: bool = False,
|
|
|
if entries:
|
|
if entries:
|
|
|
newest = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime(max(e[2] for e in entries) / 1e9))
|
|
newest = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime(max(e[2] for e in entries) / 1e9))
|
|
|
_unm = [e[0] for e in entries if e[0] not in mset]
|
|
_unm = [e[0] for e in entries if e[0] not in mset]
|
|
|
|
|
+ # ── 消费者口径再判一次(步骤 C): 扩展名"像数据"但没人读 ⇒ 缺口; 其余 ⇒ 信息 ──
|
|
|
|
|
+ # ★判据必须落在"**全部文件里没被消费者读的**"上, 不能只看 unmatched(白名单没命中的那些) ——
|
|
|
|
|
+ # 2026-09-20 实逮: `WTG01-B2.csv` 是 `*.csv`,白名单命中了它(所以 unmatched 里没有它),
|
|
|
|
|
+ # 但取数层只认 `WTG01.csv` ⇒ 只看 unmatched 就永远发现不了这类"同名不同款"的静默漏读。
|
|
|
|
|
+ spec = CONSUME.get(d.name) or {}
|
|
|
|
|
+ cg = tuple(spec.get('consume') or ())
|
|
|
|
|
+ dex = tuple(spec.get('data_exts') or ())
|
|
|
|
|
+ consumed: set[str] = set()
|
|
|
|
|
+ for g in cg:
|
|
|
|
|
+ consumed |= {p.relative_to(d).as_posix() for p in d.rglob(g) if p.is_file()}
|
|
|
|
|
+ not_consumed = [r for r, _s, _m in entries if r not in consumed]
|
|
|
|
|
+ gaps = [r for r in not_consumed if pathlib.PurePosixPath(r).suffix.lower() in dex]
|
|
|
|
|
+ others = [r for r in not_consumed if r not in gaps]
|
|
|
return dict(files=len(entries), bytes=sum(e[1] for e in entries), newest=newest,
|
|
return dict(files=len(entries), bytes=sum(e[1] for e in entries), newest=newest,
|
|
|
matched=len(mset), dirs=len(dirs), unmatched=_unm[:10], unmatched_n=len(_unm),
|
|
matched=len(mset), dirs=len(dirs), unmatched=_unm[:10], unmatched_n=len(_unm),
|
|
|
|
|
+ consumed=len(consumed), consume_how=spec.get('how', ''),
|
|
|
|
|
+ gaps=gaps[:10], gaps_n=len(gaps), others=others[:10], others_n=len(others),
|
|
|
digest=_digest(entries, deep, d, all_files, deep_max, dirs),
|
|
digest=_digest(entries, deep, d, all_files, deep_max, dirs),
|
|
|
top=[e[0] for e in sorted(entries, key=lambda x: -x[2])[:3]])
|
|
top=[e[0] for e in sorted(entries, key=lambda x: -x[2])[:3]])
|
|
|
|
|
|
|
@@ -175,11 +217,14 @@ def scan(raw_root: pathlib.Path, deep: bool = False, station_dir: pathlib.Path |
|
|
|
rel = p.relative_to(raw_root).as_posix()
|
|
rel = p.relative_to(raw_root).as_posix()
|
|
|
unclassified.append(dict(dir=rel, files=1,
|
|
unclassified.append(dict(dir=rel, files=1,
|
|
|
note='不在任何已知族的目录里 —— 没有消费者; 是新的输入族就得给它配摄入器'))
|
|
note='不在任何已知族的目录里 —— 没有消费者; 是新的输入族就得给它配摄入器'))
|
|
|
- # 已知族目录内**白名单没命中**的文件: 能进摄入的才计数, 这些是"放了但没人读"的
|
|
|
|
|
- stray = []
|
|
|
|
|
|
|
+ # 已知族目录内**消费者不会读**的件 (步骤 C): 两级 —— 缺口级(gaps, 扩展名像数据) 与 信息级(others)
|
|
|
|
|
+ stray, gaps = [], []
|
|
|
for name, fp in fam.items():
|
|
for name, fp in fam.items():
|
|
|
- if fp.get('unmatched_n'):
|
|
|
|
|
- stray.append(dict(family=name, n=fp['unmatched_n'], example=fp.get('unmatched')[:5]))
|
|
|
|
|
|
|
+ if fp.get('others_n'):
|
|
|
|
|
+ stray.append(dict(family=name, n=fp['others_n'], example=fp.get('others')[:5]))
|
|
|
|
|
+ if fp.get('gaps_n'):
|
|
|
|
|
+ gaps.append(dict(family=name, n=fp['gaps_n'], example=fp.get('gaps')[:8],
|
|
|
|
|
+ how=fp.get('consume_how', '')))
|
|
|
# 场站目录下**新增的未知子目录**(整目录级; 其文件已在上面逐件登记, 这里给个汇总便于人读)
|
|
# 场站目录下**新增的未知子目录**(整目录级; 其文件已在上面逐件登记, 这里给个汇总便于人读)
|
|
|
extra_dirs = []
|
|
extra_dirs = []
|
|
|
if station.is_dir():
|
|
if station.is_dir():
|
|
@@ -189,7 +234,8 @@ def scan(raw_root: pathlib.Path, deep: bool = False, station_dir: pathlib.Path |
|
|
|
files=sum(1 for p in sub.rglob('*') if p.is_file()),
|
|
files=sum(1 for p in sub.rglob('*') if p.is_file()),
|
|
|
note='场站目录下的新子目录(族表里没有)'))
|
|
note='场站目录下的新子目录(族表里没有)'))
|
|
|
return dict(at=time.strftime('%Y-%m-%d %H:%M:%S'), root=str(raw_root), station=str(station),
|
|
return dict(at=time.strftime('%Y-%m-%d %H:%M:%S'), root=str(raw_root), station=str(station),
|
|
|
- deep=deep, families=fam, unclassified=unclassified + extra_dirs, stray_in_families=stray)
|
|
|
|
|
|
|
+ deep=deep, families=fam, unclassified=unclassified + extra_dirs,
|
|
|
|
|
+ stray_in_families=stray, gaps=gaps)
|
|
|
|
|
|
|
|
|
|
|
|
|
def diff(old: dict, new: dict) -> list[dict]:
|
|
def diff(old: dict, new: dict) -> list[dict]:
|
|
@@ -218,7 +264,15 @@ def diff(old: dict, new: dict) -> list[dict]:
|
|
|
for u in new.get('unclassified') or []:
|
|
for u in new.get('unclassified') or []:
|
|
|
out.append(dict(family=u['dir'], kind='未归类(不在任何已知族的目录里)', files=u['files'],
|
|
out.append(dict(family=u['dir'], kind='未归类(不在任何已知族的目录里)', files=u['files'],
|
|
|
steps=[], note=u.get('note') or '没有已知消费者 ⇒ 先认领'))
|
|
steps=[], note=u.get('note') or '没有已知消费者 ⇒ 先认领'))
|
|
|
- # ★族内"白名单外"的件**不计入变化**(信息而已): 实测族目录里本来就有大量正常但不被摄入器读的东西
|
|
|
|
|
|
|
+ # ★缺口级 (步骤 C, **计入变化**): 扩展名看着是数据、但按消费者口径没人会读 ⇒ 重算也不会纳入。
|
|
|
|
|
+ # 典型: scada_10min/ 里的 `WTG01-B2.csv`(856 列 / 2026-08 数据) —— 取数层只认 `WTG01.csv`。
|
|
|
|
|
+ for g in new.get('gaps') or []:
|
|
|
|
|
+ out.append(dict(family=f'{g["family"]}(数据没人读·缺口)', kind=f'{g["n"]} 件数据不会被摄入',
|
|
|
|
|
+ files=g['n'], steps=[],
|
|
|
|
|
+ note=f'消费者口径: {g["how"]} ⇒ 例: ' + '、'.join(g['example'])
|
|
|
|
|
+ + ' —— 这些件**不会进任何产物**(重算也不会变); 要么按消费者口径改名/转换, '
|
|
|
|
|
+ '要么给取数层加"同台多件"的合并口径'))
|
|
|
|
|
+ # ★附件类(信息级,**不计入变化**): 实测族目录里本来就有大量正常但不该被摄入的东西
|
|
|
# (故障记录里的 .rar/.jpg、windcms 厂家报告 PDF、技术资料里的厂商软件 .exe/.lp …)。
|
|
# (故障记录里的 .rar/.jpg、windcms 厂家报告 PDF、技术资料里的厂商软件 .exe/.lp …)。
|
|
|
# 它们增删本身已经反映在 files/dirs/digest 上 ⇒ 该报的变化照报, 这里不再重复判一次。
|
|
# 它们增删本身已经反映在 files/dirs/digest 上 ⇒ 该报的变化照报, 这里不再重复判一次。
|
|
|
return out
|
|
return out
|
|
@@ -299,8 +353,14 @@ def main() -> int:
|
|
|
print(' 未归类(族表里没有,先认领):')
|
|
print(' 未归类(族表里没有,先认领):')
|
|
|
for u in now['unclassified']:
|
|
for u in now['unclassified']:
|
|
|
print(f' {u["dir"]:16s} {u["files"]:6d} 件 {u.get("note", "没有已知消费者")}')
|
|
print(f' {u["dir"]:16s} {u["files"]:6d} 件 {u.get("note", "没有已知消费者")}')
|
|
|
|
|
+ if now.get('gaps'):
|
|
|
|
|
+ print(' ★[缺口] 族目录里有**看着是数据、但没有任何消费者会读**的件(重算不会纳入它们):')
|
|
|
|
|
+ for g in now['gaps']:
|
|
|
|
|
+ print(f' {g["family"]:16s} {g["n"]:5d} 件 例: ' + '、'.join(g['example'][:4])
|
|
|
|
|
+ + (' …' if g['n'] > 4 else ''))
|
|
|
|
|
+ print(f' 消费者口径: {g["how"]}')
|
|
|
if now.get('stray_in_families'):
|
|
if now.get('stray_in_families'):
|
|
|
- print(' (信息,不计入"变化")族目录里**摄入白名单没命中**的件 —— 摄入器不会读它们:')
|
|
|
|
|
|
|
+ print(' (信息,不计入"变化")族目录里**消费者不读的附件类**件:')
|
|
|
for s in now['stray_in_families']:
|
|
for s in now['stray_in_families']:
|
|
|
print(f' {s["family"]:16s} {s["n"]:5d} 件 例: ' + '、'.join(s['example'][:3]))
|
|
print(f' {s["family"]:16s} {s["n"]:5d} 件 例: ' + '、'.join(s['example'][:3]))
|
|
|
|
|
|
|
@@ -314,8 +374,13 @@ def main() -> int:
|
|
|
print(f' 最近落盘: ' + '、'.join(c['top']))
|
|
print(f' 最近落盘: ' + '、'.join(c['top']))
|
|
|
if c.get('steps'):
|
|
if c.get('steps'):
|
|
|
print(f' 该跑: ' + ' / '.join(c['steps']))
|
|
print(f' 该跑: ' + ' / '.join(c['steps']))
|
|
|
- print('\n 建议: 跑一次重算把新数据纳入产物 —— python scripts/rebuild_all.py'
|
|
|
|
|
- '(默认全跑;若你用了 --skip-*,请确认没跳过上面列出的步)')
|
|
|
|
|
|
|
+ _real = [c for c in changes if '缺口' not in c['family']]
|
|
|
|
|
+ if _real:
|
|
|
|
|
+ print('\n 建议: 跑一次重算把新数据纳入产物 —— python scripts/rebuild_all.py'
|
|
|
|
|
+ '(默认全跑;若你用了 --skip-*,请确认没跳过上面列出的步)')
|
|
|
|
|
+ if len(_real) < len(changes):
|
|
|
|
|
+ print(' ★缺口类变化**重算也纳入不了**(消费者根本不读这些件): 先按上面写的口径改名/转换, '
|
|
|
|
|
+ '或给取数层加"同台多件"的合并口径(见 docs/输入数据放置指导_v0.1.md §4)')
|
|
|
else:
|
|
else:
|
|
|
print('\n== 与上次快照一致,没有新数据 ==' if old else '\n== 还没有快照(跑 --write 记一次基线)==')
|
|
print('\n== 与上次快照一致,没有新数据 ==' if old else '\n== 还没有快照(跑 --write 记一次基线)==')
|
|
|
|
|
|