Selaa lähdekoodia

步骤 C: 让"放了没人读的数据"当场可见 (用户令 2026-09-20)

背景 (实测): data/raw/如东/scada_10min/ 里新来了 38 件 `WTG01-B2.csv…WTG38-B2.csv`
(856 列 / 2026-08-01~09-01 数据), 而取数层只认 `<台号>.csv`(WTG01.csv) ⇒ 这批数据**从未被读过**;
旧扫描只看"扩展名白名单", `*.csv` 命中了它 ⇒ 既算作进了摄入, 又不会出现在任何"没人读"的提示里
⇒ 重算跑完页面时间窗仍止于 2026-07, 而系统一声不响。

改动 (scripts/raw_scan.py):
- 新增 CONSUME 表: 逐族写清**消费者真实取数口径**(如 scada_10min 只认 `WTG??.csv`) + 哪些扩展名算"数据件"
- 判据从"白名单没命中"改成"**全部文件里没被消费者读的**"(只看白名单会永远发现不了 `WTG01-B2.csv`
  这种"白名单命中、消费者不认"的静默漏读)
- 两级输出: **缺口级**(像数据却没人读) 计入变化(rc=4) + ★[缺口] 块大声报出消费者口径与件例;
  **信息级**(附件/压缩包/图片/厂商软件) 只列信息不计变化
- 建议行分流: 只有缺口时明确说"重算也纳入不了, 先改名/转换或给取数层加同台多件口径"

实测: scada_10min 报 `38 件数据不会被摄入` + 消费者口径; rc=4; 附件类仍只在信息块。
zhouyang.xie 3 viikkoa sitten
vanhempi
commit
dd9f727590
1 muutettua tiedostoa jossa 74 lisäystä ja 9 poistoa
  1. 74 9
      scripts/raw_scan.py

+ 74 - 9
scripts/raw_scan.py

@@ -79,6 +79,33 @@ TECH_NOTE = '厂商技术资料(故障处理手册/维护 WI/图纸/对译表
 # raw 根下**随包发的说明件**(不是数据): 别把它们报成"未归类"。口径: 只有名字在白名单里且不参与摄入。
 IGNORE_FILES = {'README_把原始数据放这里.txt'}
 
+# ── 消费者取数口径(用户令 2026-09-20 步骤 C:让"放了没人读"当场可见)────────────────────────
+# ★为什么必须有它:族表原来只按**扩展名白名单**计件 —— 往 `scada_10min/` 里放 `WTG01-B2.csv`(856 列、
+#   2026-08 数据、与 `<台号>.csv` 命名不同)时,它命中 `*.csv` ⇒ 既算"进了摄入",又不会出现在任何
+#   "没人读"的提示里;而取数层只认 `<台号>.csv` ⇒ **这批数据从来没被读过**,重算后时间窗自然不变。
+#   本表按**消费者真实取数口径**再判一次,并把结果分成两级:
+#     · 缺口级 (gate):看着是数据(扩展名在 data_exts 里) 但没有任何消费者会读 ⇒ **计入变化(rc=4) + 大声报**
+#     · 信息级 (info):附件/压缩包/图片/厂商软件这类,本来就不该被摄入 ⇒ 只列信息,不计变化
+CONSUME: dict[str, dict] = {
+    '故障报警': dict(consume=('*.xls',), data_exts=('.xls', '.xlsx', '.csv'),
+                     how='摄入器读 *.xls 的 XML 事件导出(TimeOn/Alarmcode)'),
+    '风机故障记录': dict(consume=('*.xls', '*.xlsx'), data_exts=('.xls', '.xlsx', '.csv'),
+                         how='台账读取 *.xls*(rar/jpg 是附件,不摄入)'),
+    '油样报告': dict(consume=('*.pdf',), data_exts=('.pdf',), how='油样摄取读 *.pdf'),
+    'scada_10min': dict(consume=('WTG??.csv',), data_exts=('.csv', '.mdb', '.zip'),
+                        how='取数层**按台号精确取 `<台号>.csv`**(WTG01.csv);别的命名不会被读'),
+    'scada_1min': dict(consume=('???.csv',), data_exts=('.csv', '.mdb', '.zip'),
+                       how='取数层按内部台号取 `<台号>.csv`(01E.csv)'),
+    'scada_mdb': dict(consume=('*-10min.mdb', '*-1min.mdb', '*.zip'), data_exts=('.mdb', '.zip'),
+                      how='取数层读取按月的库(2026-08-10min.mdb);CSV 缺失时才回落'),
+    'windcms': dict(consume=('*_decode.json', '*.pdf', '*.docx', '*.md', '*.json'),
+                    data_exts=('.json', '.zip', '.csv', '.mdb'),
+                    how='振动摄入读 *_decode.json;厂家报告 PDF 由报告转录侧读'),
+    'm5_cms_tcm': dict(consume=('*.json', '*.docx', '*.pdf', '*.md'), data_exts=('.json', '.zip'),
+                       how='现场正本 handoff/component_history 与厂家报告'),
+    TECH_DIR: dict(consume=('*',), data_exts=(), how='本体读整棵树(白名单=全部)'),
+}
+
 
 def _digest(entries: list[tuple[str, int, int]], deep: bool, base: pathlib.Path, files: list[pathlib.Path],
             deep_max: int, dirs: list[str] | None = None) -> str:
@@ -138,8 +165,23 @@ def fingerprint(d: pathlib.Path, patterns: tuple[str, ...], deep: bool = False,
     if entries:
         newest = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime(max(e[2] for e in entries) / 1e9))
     _unm = [e[0] for e in entries if e[0] not in mset]
+    # ── 消费者口径再判一次(步骤 C): 扩展名"像数据"但没人读 ⇒ 缺口; 其余 ⇒ 信息 ──
+    # ★判据必须落在"**全部文件里没被消费者读的**"上, 不能只看 unmatched(白名单没命中的那些) ——
+    #   2026-09-20 实逮: `WTG01-B2.csv` 是 `*.csv`,白名单命中了它(所以 unmatched 里没有它),
+    #   但取数层只认 `WTG01.csv` ⇒ 只看 unmatched 就永远发现不了这类"同名不同款"的静默漏读。
+    spec = CONSUME.get(d.name) or {}
+    cg = tuple(spec.get('consume') or ())
+    dex = tuple(spec.get('data_exts') or ())
+    consumed: set[str] = set()
+    for g in cg:
+        consumed |= {p.relative_to(d).as_posix() for p in d.rglob(g) if p.is_file()}
+    not_consumed = [r for r, _s, _m in entries if r not in consumed]
+    gaps = [r for r in not_consumed if pathlib.PurePosixPath(r).suffix.lower() in dex]
+    others = [r for r in not_consumed if r not in gaps]
     return dict(files=len(entries), bytes=sum(e[1] for e in entries), newest=newest,
                 matched=len(mset), dirs=len(dirs), unmatched=_unm[:10], unmatched_n=len(_unm),
+                consumed=len(consumed), consume_how=spec.get('how', ''),
+                gaps=gaps[:10], gaps_n=len(gaps), others=others[:10], others_n=len(others),
                 digest=_digest(entries, deep, d, all_files, deep_max, dirs),
                 top=[e[0] for e in sorted(entries, key=lambda x: -x[2])[:3]])
 
@@ -175,11 +217,14 @@ def scan(raw_root: pathlib.Path, deep: bool = False, station_dir: pathlib.Path |
         rel = p.relative_to(raw_root).as_posix()
         unclassified.append(dict(dir=rel, files=1,
                                  note='不在任何已知族的目录里 —— 没有消费者; 是新的输入族就得给它配摄入器'))
-    # 已知族目录内**白名单没命中**的文件: 能进摄入的才计数, 这些是"放了但没人读"的
-    stray = []
+    # 已知族目录内**消费者不会读**的件 (步骤 C): 两级 —— 缺口级(gaps, 扩展名像数据) 与 信息级(others)
+    stray, gaps = [], []
     for name, fp in fam.items():
-        if fp.get('unmatched_n'):
-            stray.append(dict(family=name, n=fp['unmatched_n'], example=fp.get('unmatched')[:5]))
+        if fp.get('others_n'):
+            stray.append(dict(family=name, n=fp['others_n'], example=fp.get('others')[:5]))
+        if fp.get('gaps_n'):
+            gaps.append(dict(family=name, n=fp['gaps_n'], example=fp.get('gaps')[:8],
+                             how=fp.get('consume_how', '')))
     # 场站目录下**新增的未知子目录**(整目录级; 其文件已在上面逐件登记, 这里给个汇总便于人读)
     extra_dirs = []
     if station.is_dir():
@@ -189,7 +234,8 @@ def scan(raw_root: pathlib.Path, deep: bool = False, station_dir: pathlib.Path |
                                        files=sum(1 for p in sub.rglob('*') if p.is_file()),
                                        note='场站目录下的新子目录(族表里没有)'))
     return dict(at=time.strftime('%Y-%m-%d %H:%M:%S'), root=str(raw_root), station=str(station),
-                deep=deep, families=fam, unclassified=unclassified + extra_dirs, stray_in_families=stray)
+                deep=deep, families=fam, unclassified=unclassified + extra_dirs,
+                stray_in_families=stray, gaps=gaps)
 
 
 def diff(old: dict, new: dict) -> list[dict]:
@@ -218,7 +264,15 @@ def diff(old: dict, new: dict) -> list[dict]:
     for u in new.get('unclassified') or []:
         out.append(dict(family=u['dir'], kind='未归类(不在任何已知族的目录里)', files=u['files'],
                         steps=[], note=u.get('note') or '没有已知消费者 ⇒ 先认领'))
-    # ★族内"白名单外"的件**不计入变化**(信息而已): 实测族目录里本来就有大量正常但不被摄入器读的东西
+    # ★缺口级 (步骤 C, **计入变化**): 扩展名看着是数据、但按消费者口径没人会读 ⇒ 重算也不会纳入。
+    #   典型: scada_10min/ 里的 `WTG01-B2.csv`(856 列 / 2026-08 数据) —— 取数层只认 `WTG01.csv`。
+    for g in new.get('gaps') or []:
+        out.append(dict(family=f'{g["family"]}(数据没人读·缺口)', kind=f'{g["n"]} 件数据不会被摄入',
+                        files=g['n'], steps=[],
+                        note=f'消费者口径: {g["how"]} ⇒ 例: ' + '、'.join(g['example'])
+                             + ' —— 这些件**不会进任何产物**(重算也不会变); 要么按消费者口径改名/转换, '
+                               '要么给取数层加"同台多件"的合并口径'))
+    # ★附件类(信息级,**不计入变化**): 实测族目录里本来就有大量正常但不该被摄入的东西
     #   (故障记录里的 .rar/.jpg、windcms 厂家报告 PDF、技术资料里的厂商软件 .exe/.lp …)。
     #   它们增删本身已经反映在 files/dirs/digest 上 ⇒ 该报的变化照报, 这里不再重复判一次。
     return out
@@ -299,8 +353,14 @@ def main() -> int:
         print('  未归类(族表里没有,先认领):')
         for u in now['unclassified']:
             print(f'    {u["dir"]:16s} {u["files"]:6d} 件  {u.get("note", "没有已知消费者")}')
+    if now.get('gaps'):
+        print('  ★[缺口] 族目录里有**看着是数据、但没有任何消费者会读**的件(重算不会纳入它们):')
+        for g in now['gaps']:
+            print(f'    {g["family"]:16s} {g["n"]:5d} 件  例: ' + '、'.join(g['example'][:4])
+                  + (' …' if g['n'] > 4 else ''))
+            print(f'      消费者口径: {g["how"]}')
     if now.get('stray_in_families'):
-        print('  (信息,不计入"变化")族目录里**摄入白名单没命中**的件 —— 摄入器不会读它们:')
+        print('  (信息,不计入"变化")族目录里**消费者不读的附件类**件:')
         for s in now['stray_in_families']:
             print(f'    {s["family"]:16s} {s["n"]:5d} 件  例: ' + '、'.join(s['example'][:3]))
 
@@ -314,8 +374,13 @@ def main() -> int:
                 print(f'      最近落盘: ' + '、'.join(c['top']))
             if c.get('steps'):
                 print(f'      该跑: ' + ' / '.join(c['steps']))
-        print('\n  建议: 跑一次重算把新数据纳入产物 —— python scripts/rebuild_all.py'
-              '(默认全跑;若你用了 --skip-*,请确认没跳过上面列出的步)')
+        _real = [c for c in changes if '缺口' not in c['family']]
+        if _real:
+            print('\n  建议: 跑一次重算把新数据纳入产物 —— python scripts/rebuild_all.py'
+                  '(默认全跑;若你用了 --skip-*,请确认没跳过上面列出的步)')
+        if len(_real) < len(changes):
+            print('  ★缺口类变化**重算也纳入不了**(消费者根本不读这些件): 先按上面写的口径改名/转换, '
+                  '或给取数层加"同台多件"的合并口径(见 docs/输入数据放置指导_v0.1.md §4)')
     else:
         print('\n== 与上次快照一致,没有新数据 ==' if old else '\n== 还没有快照(跑 --write 记一次基线)==')