|
|
@@ -188,7 +188,8 @@ def verify(out: pathlib.Path, cls_filter: str | None, refresh: bool, deep: bool)
|
|
|
src = src_month_rows(cls_filter, out / '_csv_month_rows.json', refresh) if deep else {}
|
|
|
print(f'== CSV → MDB 全量核对 · {out} ==')
|
|
|
print(f' {"类":6s} {"月份":8s} {"mdb 各表行数":18s} {"应得行数":10s} 判定 zip')
|
|
|
- bad = []
|
|
|
+ bad, short, over, minor = [], [], [], []
|
|
|
+ BOUNDARY = 0.005 # 超日历 ≤0.5% 视为**月末边界行**(有些台次月 00:00 那一行落在本月)
|
|
|
for m in mdbs:
|
|
|
cls = m.stem.rsplit('-', 1)[1]
|
|
|
ym = '-'.join(m.stem.split('-')[:2])
|
|
|
@@ -199,20 +200,37 @@ def verify(out: pathlib.Path, cls_filter: str | None, refresh: bool, deep: bool)
|
|
|
want = CAD.get(cls, 0) * calendar.monthrange(y, mo)[1] * 38
|
|
|
aligned = len(set(rows)) == 1 if rows else False
|
|
|
d = (rows[0] - want) if rows else -want
|
|
|
- # 日历模式: 少于日历=数据缺口(正常), 多于日历才是错; 深核对模式: 必须与源逐行相等
|
|
|
- ok = bool(rows) and aligned and (rows[0] == want if deep else rows[0] <= want)
|
|
|
- tag = ('OK' if ok else '异常') if deep else ('一致' if d == 0 else (f'缺口{d:+d}' if d < 0 else f'超日历{d:+d}'))
|
|
|
+ # 深核对: 必须与源逐行相等; 日历模式: 少行=该月数据本身不齐(正常), 多行超过 0.5% 才算转换错
|
|
|
+ if deep:
|
|
|
+ ok = bool(rows) and aligned and rows[0] == want
|
|
|
+ else:
|
|
|
+ ok = bool(rows) and aligned and d <= want * BOUNDARY
|
|
|
+ tag = (('OK' if ok else '异常') if deep else
|
|
|
+ ('一致' if d == 0 else (f'缺口{d:+d}' if d < 0 else
|
|
|
+ (f'超日历{d:+d}' if d > want * BOUNDARY else f'边界{d:+d}'))))
|
|
|
zp = m.parent / f'{ym}-{cls}.zip'
|
|
|
ztxt = f'有 {zp.stat().st_size / 1e6:.1f} MB' if zp.is_file() else '无'
|
|
|
print(f' {cls:6s} {ym:8s} {",".join(map(str, rows)):18s} {want:<10d} '
|
|
|
f'{tag:12s} {ztxt}')
|
|
|
if not ok:
|
|
|
bad.append((cls, ym, rows, want))
|
|
|
- print(f' 合计: {len(mdbs)} 个月库 · 一致 {len(mdbs) - len(bad)} · 差异 {len(bad)}'
|
|
|
- + ('' if deep else '(按日历推算; 差异多为该月首尾不完整, 属正常 ⇒ 要按源逐行核对加 --deep)'))
|
|
|
+ if rows and d < 0:
|
|
|
+ short.append((cls, ym, rows, want))
|
|
|
+ elif rows and d > 0:
|
|
|
+ (over if d > want * BOUNDARY else minor).append((cls, ym, rows, want))
|
|
|
+ # ★2026-09-19 修: 原来汇总只报"一致 N / 差异 K", 而日历判据是 `rows <= want` ⇒ **任何缺口都算一致**,
|
|
|
+ # 于是 1min 那张表有 8 个月短几百到一万多行, 汇总却是"一致 18 · 差异 0"(自相矛盾, 只有看表才发现)。
|
|
|
+ # 现在四档分开报: 一致 / 缺口(少, 数据本身) / 边界(多但 ≤0.5%, 月末那一行) / 超日历(多且超 0.5%, 才是转换错)。
|
|
|
+ print(f' 合计: {len(mdbs)} 个月库 · 一致 {len(mdbs) - len(bad)} · 缺口 {len(short)} · '
|
|
|
+ f'边界 {len(minor)} · 超日历 {len(over)}'
|
|
|
+ + ('' if deep else '(按日历推算; 少行=该月数据本身不齐, 多行>0.5% 才是转换错; 逐行对源加 --deep)'))
|
|
|
+ for cls, ym, rows, want in sorted(short, key=lambda x: x[2][0] - x[3])[:3]:
|
|
|
+ print(f' [i] 缺口最大: {cls} {ym}: {rows[0]:,} 行 (差 {rows[0] - want:+,} = '
|
|
|
+ f'{100 * (rows[0] - want) / want:.2f}%)')
|
|
|
for cls, ym, rows, want in bad[:8]:
|
|
|
print(f' [i] {cls} {ym}: mdb={rows} vs 应得 {want}')
|
|
|
- print(' 结论: ' + ('全部一致(各分片表行数相同 ⇒ 逐行对齐)' if not bad else '有差异, 见上'))
|
|
|
+ print(' 结论: ' + ('全部一致(各分片表行数相同 ⇒ 逐行对齐)' if not bad else
|
|
|
+ ('有差异, 见上' if over or deep else '分片表行数整齐; 少行属当月数据不齐(见上), 无转换错')))
|
|
|
return 0 if not bad else 5
|
|
|
|
|
|
|