|
@@ -40,6 +40,7 @@ import hashlib
|
|
|
import json
|
|
import json
|
|
|
import os
|
|
import os
|
|
|
import pathlib
|
|
import pathlib
|
|
|
|
|
+import re
|
|
|
import sys
|
|
import sys
|
|
|
import time
|
|
import time
|
|
|
|
|
|
|
@@ -78,6 +79,12 @@ TECH_STEPS = ('⑦ 本体: 码表/手册/文档',)
|
|
|
TECH_NOTE = '厂商技术资料(故障处理手册/维护 WI/图纸/对译表)→ 本体对象库/检索索引/实机参数表'
|
|
TECH_NOTE = '厂商技术资料(故障处理手册/维护 WI/图纸/对译表)→ 本体对象库/检索索引/实机参数表'
|
|
|
# raw 根下**随包发的说明件**(不是数据): 别把它们报成"未归类"。口径: 只有名字在白名单里且不参与摄入。
|
|
# raw 根下**随包发的说明件**(不是数据): 别把它们报成"未归类"。口径: 只有名字在白名单里且不参与摄入。
|
|
|
IGNORE_FILES = {'README_把原始数据放这里.txt'}
|
|
IGNORE_FILES = {'README_把原始数据放这里.txt'}
|
|
|
|
|
+# ★2026-09-21 用户令「按月 10min 提取件就放在 <场站>/ 下」: 它是**给人核对/交接的导出件**, 不是摄入源
|
|
|
|
|
+# (摄入走 `<台号>.csv` + 同台补充件)。不加这条, 每次扫描都会把它报成"未归类 + 一处变化" ⇒
|
|
|
|
|
+# `--auto` / raw_watch 会永远以为"来了新数据", 每轮都触发一次全量重算。
|
|
|
|
|
+IGNORE_PATTERNS: tuple[tuple[str, str], ...] = (
|
|
|
|
|
+ (r'^scada_\d+min_\d{6}\.csv$', '按月提取/核对件(非摄入源;摄入走 <台号>.csv + 同台补充件)'),
|
|
|
|
|
+)
|
|
|
|
|
|
|
|
# ── 消费者取数口径(用户令 2026-09-20 步骤 C:让"放了没人读"当场可见)────────────────────────
|
|
# ── 消费者取数口径(用户令 2026-09-20 步骤 C:让"放了没人读"当场可见)────────────────────────
|
|
|
# ★为什么必须有它:族表原来只按**扩展名白名单**计件 —— 往 `scada_10min/` 里放 `WTG01-B2.csv`(856 列、
|
|
# ★为什么必须有它:族表原来只按**扩展名白名单**计件 —— 往 `scada_10min/` 里放 `WTG01-B2.csv`(856 列、
|
|
@@ -98,7 +105,15 @@ CONSUME: dict[str, dict] = {
|
|
|
'scada_1min': dict(consume=('???.csv', '???-*.csv', '???_*.csv'), data_exts=('.csv', '.mdb', '.zip'),
|
|
'scada_1min': dict(consume=('???.csv', '???-*.csv', '???_*.csv'), data_exts=('.csv', '.mdb', '.zip'),
|
|
|
how='取数层按内部台号取 `<台号>.csv`(01E.csv)+ 同台补充件'),
|
|
how='取数层按内部台号取 `<台号>.csv`(01E.csv)+ 同台补充件'),
|
|
|
'scada_mdb': dict(consume=('*-10min.mdb', '*-1min.mdb', '*.zip'), data_exts=('.mdb', '.zip'),
|
|
'scada_mdb': dict(consume=('*-10min.mdb', '*-1min.mdb', '*.zip'), data_exts=('.mdb', '.zip'),
|
|
|
- how='取数层读取按月的库(2026-08-10min.mdb);CSV 缺失时才回落'),
|
|
|
|
|
|
|
+ how='取数层读取按月的库(2026-08-10min.mdb);CSV 缺失时才回落',
|
|
|
|
|
+ # ★2026-09-21: 现场按**类目**交付的月度通道组库(7月-8月交付 zip 里的
|
|
|
|
|
+ # `2026-0X-<类>.mdb`)。它是 `scada_10min` 同台补充件的**上游**:取数层不直接读它,
|
|
|
|
|
+ # 要先转成 `scada_10min/<台号>-<YYYYMM>.csv`。故列为"上游归档"(信息级),
|
|
|
|
|
+ # 既不当缺口(不算变化/不触发重算),也不假装被摄入。
|
|
|
|
|
+ upstream=tuple(f'*-{c}.mdb' for c in
|
|
|
|
|
+ ('cnt', 'din', 'dot', 'flg', 'grd', 'int', 'prs', 'scd', 'std', 'sum', 'tmp', 'tur')),
|
|
|
|
|
+ upstream_how='现场月度**通道组**归档(上游件):先转成同台 10min 补充件 '
|
|
|
|
|
+ '`scada_10min/<台号>-<YYYYMM>.csv` 才被消费(口径见 docs/输入数据放置指导_v0.1.md §4.1)'),
|
|
|
'windcms': dict(consume=('*_decode.json', '*.pdf', '*.docx', '*.md', '*.json'),
|
|
'windcms': dict(consume=('*_decode.json', '*.pdf', '*.docx', '*.md', '*.json'),
|
|
|
data_exts=('.json', '.zip', '.csv', '.mdb'),
|
|
data_exts=('.json', '.zip', '.csv', '.mdb'),
|
|
|
how='振动摄入读 *_decode.json;厂家报告 PDF 由报告转录侧读'),
|
|
how='振动摄入读 *_decode.json;厂家报告 PDF 由报告转录侧读'),
|
|
@@ -176,12 +191,19 @@ def fingerprint(d: pathlib.Path, patterns: tuple[str, ...], deep: bool = False,
|
|
|
consumed: set[str] = set()
|
|
consumed: set[str] = set()
|
|
|
for g in cg:
|
|
for g in cg:
|
|
|
consumed |= {p.relative_to(d).as_posix() for p in d.rglob(g) if p.is_file()}
|
|
consumed |= {p.relative_to(d).as_posix() for p in d.rglob(g) if p.is_file()}
|
|
|
|
|
+ # ── 上游归档件(客户口径里"已知但不由取数层直接读"的): 归信息级, 不算缺口、不计变化 ──
|
|
|
|
|
+ upat = tuple(spec.get('upstream') or ())
|
|
|
|
|
+ upstream: set[str] = set()
|
|
|
|
|
+ for g in upat:
|
|
|
|
|
+ upstream |= {p.relative_to(d).as_posix() for p in d.rglob(g) if p.is_file()}
|
|
|
not_consumed = [r for r, _s, _m in entries if r not in consumed]
|
|
not_consumed = [r for r, _s, _m in entries if r not in consumed]
|
|
|
- gaps = [r for r in not_consumed if pathlib.PurePosixPath(r).suffix.lower() in dex]
|
|
|
|
|
|
|
+ gaps = [r for r in not_consumed
|
|
|
|
|
+ if r not in upstream and pathlib.PurePosixPath(r).suffix.lower() in dex]
|
|
|
others = [r for r in not_consumed if r not in gaps]
|
|
others = [r for r in not_consumed if r not in gaps]
|
|
|
return dict(files=len(entries), bytes=sum(e[1] for e in entries), newest=newest,
|
|
return dict(files=len(entries), bytes=sum(e[1] for e in entries), newest=newest,
|
|
|
matched=len(mset), dirs=len(dirs), unmatched=_unm[:10], unmatched_n=len(_unm),
|
|
matched=len(mset), dirs=len(dirs), unmatched=_unm[:10], unmatched_n=len(_unm),
|
|
|
consumed=len(consumed), consume_how=spec.get('how', ''),
|
|
consumed=len(consumed), consume_how=spec.get('how', ''),
|
|
|
|
|
+ upstream=len(upstream), upstream_how=spec.get('upstream_how', ''),
|
|
|
gaps=gaps[:10], gaps_n=len(gaps), others=others[:10], others_n=len(others),
|
|
gaps=gaps[:10], gaps_n=len(gaps), others=others[:10], others_n=len(others),
|
|
|
digest=_digest(entries, deep, d, all_files, deep_max, dirs),
|
|
digest=_digest(entries, deep, d, all_files, deep_max, dirs),
|
|
|
top=[e[0] for e in sorted(entries, key=lambda x: -x[2])[:3]])
|
|
top=[e[0] for e in sorted(entries, key=lambda x: -x[2])[:3]])
|
|
@@ -215,17 +237,21 @@ def scan(raw_root: pathlib.Path, deep: bool = False, station_dir: pathlib.Path |
|
|
|
continue
|
|
continue
|
|
|
if p.name in IGNORE_FILES:
|
|
if p.name in IGNORE_FILES:
|
|
|
continue # 随包的放置说明件: 不是数据, 不报"未归类"
|
|
continue # 随包的放置说明件: 不是数据, 不报"未归类"
|
|
|
|
|
+ if any(re.match(rx, p.name) for rx, _why in IGNORE_PATTERNS):
|
|
|
|
|
+ continue # 交付核对用的导出件(位置由用户指定): 非摄入源, 不报"未归类"
|
|
|
rel = p.relative_to(raw_root).as_posix()
|
|
rel = p.relative_to(raw_root).as_posix()
|
|
|
unclassified.append(dict(dir=rel, files=1,
|
|
unclassified.append(dict(dir=rel, files=1,
|
|
|
note='不在任何已知族的目录里 —— 没有消费者; 是新的输入族就得给它配摄入器'))
|
|
note='不在任何已知族的目录里 —— 没有消费者; 是新的输入族就得给它配摄入器'))
|
|
|
# 已知族目录内**消费者不会读**的件 (步骤 C): 两级 —— 缺口级(gaps, 扩展名像数据) 与 信息级(others)
|
|
# 已知族目录内**消费者不会读**的件 (步骤 C): 两级 —— 缺口级(gaps, 扩展名像数据) 与 信息级(others)
|
|
|
- stray, gaps = [], []
|
|
|
|
|
|
|
+ stray, gaps, upstream = [], [], []
|
|
|
for name, fp in fam.items():
|
|
for name, fp in fam.items():
|
|
|
if fp.get('others_n'):
|
|
if fp.get('others_n'):
|
|
|
stray.append(dict(family=name, n=fp['others_n'], example=fp.get('others')[:5]))
|
|
stray.append(dict(family=name, n=fp['others_n'], example=fp.get('others')[:5]))
|
|
|
if fp.get('gaps_n'):
|
|
if fp.get('gaps_n'):
|
|
|
gaps.append(dict(family=name, n=fp['gaps_n'], example=fp.get('gaps')[:8],
|
|
gaps.append(dict(family=name, n=fp['gaps_n'], example=fp.get('gaps')[:8],
|
|
|
how=fp.get('consume_how', '')))
|
|
how=fp.get('consume_how', '')))
|
|
|
|
|
+ if fp.get('upstream'):
|
|
|
|
|
+ upstream.append(dict(family=name, n=fp['upstream'], how=fp.get('upstream_how', '')))
|
|
|
# 场站目录下**新增的未知子目录**(整目录级; 其文件已在上面逐件登记, 这里给个汇总便于人读)
|
|
# 场站目录下**新增的未知子目录**(整目录级; 其文件已在上面逐件登记, 这里给个汇总便于人读)
|
|
|
extra_dirs = []
|
|
extra_dirs = []
|
|
|
if station.is_dir():
|
|
if station.is_dir():
|
|
@@ -236,7 +262,7 @@ def scan(raw_root: pathlib.Path, deep: bool = False, station_dir: pathlib.Path |
|
|
|
note='场站目录下的新子目录(族表里没有)'))
|
|
note='场站目录下的新子目录(族表里没有)'))
|
|
|
return dict(at=time.strftime('%Y-%m-%d %H:%M:%S'), root=str(raw_root), station=str(station),
|
|
return dict(at=time.strftime('%Y-%m-%d %H:%M:%S'), root=str(raw_root), station=str(station),
|
|
|
deep=deep, families=fam, unclassified=unclassified + extra_dirs,
|
|
deep=deep, families=fam, unclassified=unclassified + extra_dirs,
|
|
|
- stray_in_families=stray, gaps=gaps)
|
|
|
|
|
|
|
+ stray_in_families=stray, gaps=gaps, upstream=upstream)
|
|
|
|
|
|
|
|
|
|
|
|
|
def diff(old: dict, new: dict) -> list[dict]:
|
|
def diff(old: dict, new: dict) -> list[dict]:
|
|
@@ -360,6 +386,11 @@ def main() -> int:
|
|
|
print(f' {g["family"]:16s} {g["n"]:5d} 件 例: ' + '、'.join(g['example'][:4])
|
|
print(f' {g["family"]:16s} {g["n"]:5d} 件 例: ' + '、'.join(g['example'][:4])
|
|
|
+ (' …' if g['n'] > 4 else ''))
|
|
+ (' …' if g['n'] > 4 else ''))
|
|
|
print(f' 消费者口径: {g["how"]}')
|
|
print(f' 消费者口径: {g["how"]}')
|
|
|
|
|
+ if now.get('upstream'):
|
|
|
|
|
+ print(' (信息,不计入"变化")族目录里的**上游归档件**(要先转换才被消费):')
|
|
|
|
|
+ for u in now['upstream']:
|
|
|
|
|
+ print(f' {u["family"]:16s} {u["n"]:5d} 件')
|
|
|
|
|
+ print(f' {u["how"]}')
|
|
|
if now.get('stray_in_families'):
|
|
if now.get('stray_in_families'):
|
|
|
print(' (信息,不计入"变化")族目录里**消费者不读的附件类**件:')
|
|
print(' (信息,不计入"变化")族目录里**消费者不读的附件类**件:')
|
|
|
for s in now['stray_in_families']:
|
|
for s in now['stray_in_families']:
|