Explorar el Código

csv_to_mdb: 装载后 COM 收尾失败不再吞掉整月 + 补 zip 模式

全量转换实逮 (2026-09-18): 1min 2025-07 / 2025-09 两个月 Powershell 非零退出, 而 stdout 里每张分片表的 tN=<行数> 都打出来了(且与切片行数相符) —— 数据其实已装完, 失败在 .Close()/.Quit() 的 COM 收尾。原判失败 ⇒ 整月当没跑成: 不打印 + 行, 也不写 zip, 1.3 GB 的 .mdb 白躺在盘上(每月 ~10 分钟抽行+装入)。

- to_mdb: 判据改成看事实 —— nt 张表都报出行数就算装成, 同时把 rc 与 stderr 尾巴响亮打出来(不静默)。
- 新增 --rezip: 只给在位但缺 zip 的月库补 zip(读 .mdb, 不重算)。
zhouyang.xie hace 3 semanas
padre
commit
4afb122a22
Se han modificado 1 ficheros con 38 adiciones y 3 borrados
  1. 38 3
      scripts/csv_to_mdb.py

+ 38 - 3
scripts/csv_to_mdb.py

@@ -112,10 +112,21 @@ def to_mdb(db: pathlib.Path, tmp: pathlib.Path, nt: int) -> list[str]:
     ps.write_text('\n'.join(L), encoding='utf-8-sig')
     r = subprocess.run(['powershell', '-NoProfile', '-ExecutionPolicy', 'Bypass', '-File', str(ps)],
                        capture_output=True, text=True, errors='replace', timeout=7200)
+    got = [x.strip() for x in (r.stdout or '').strip().splitlines() if x.strip()]
+    counts = [x for x in got if re.fullmatch(r't\d+=\d+', x)]
     if r.returncode != 0:
-        print('  [X] ' + (r.stderr or r.stdout or '').strip()[:240])
-        return []
-    return [x.strip() for x in (r.stdout or '').strip().splitlines() if x.strip()]
+        # ★2026-09-18 实逮: 全量转换里 1min 2025-07 / 2025-09 两个月退出码非 0, 而**数据其实已经装完** ——
+        #   stdout 里每张表的 `tN=<行数>` 都打出来了(且与切片行数逐月相符), 失败发生在装载之后的
+        #   COM 收尾 ($c.Close()/$acc.Quit() 那一步偶发 "未指定的错误")。原先这里直接判失败 ⇒
+        #   整月被当成没跑成: 不打印 `+ …mdb`, 也不写 zip, 而 1.3 GB 的 .mdb 已经在盘上躺着
+        #   (每月的抽行+装入 ≈ 10 分钟, 白扔)。判据改成**看事实**: 所有分片表都报出了行数 ⇒ 算成功,
+        #   但把退出码与 stderr 尾巴**响亮打出来**(不静默), 让人知道收尾那步出过岔子。
+        if len(counts) < nt:
+            print('  [X] ' + (r.stderr or r.stdout or '').strip()[:240])
+            return []
+        print(f'  [!] PowerShell 收尾非零退出 (rc={r.returncode}), 但 {len(counts)}/{nt} 张分片表都报出了行数 '
+              f'⇒ 按"已装成"处理; 尾巴: ' + (r.stderr or '').strip().replace('\n', ' ')[:160])
+    return counts
 
 
 def mdb_rows(mdbs: list[pathlib.Path]) -> dict:
@@ -205,6 +216,27 @@ def verify(out: pathlib.Path, cls_filter: str | None, refresh: bool, deep: bool)
     return 0 if not bad else 5
 
 
+def rezip(out: pathlib.Path, cls: str) -> int:
+    """月库在位但 zip 缺失 → 只补 zip (不重算, 不碰 .mdb)。
+
+    为什么需要: 2026-09-18 全量转换实测 1min 2025-07 / 2025-09 两个月"装载后收尾失败" ⇒ 当月的
+    zip 没写出来(见 to_mdb 的说明), 而 .mdb 是好的。重算一个月要 ~10 分钟, 补个 zip 只要几秒;
+    归档形态(现场是年度 zip)又不能缺, 故单列一个只读 .mdb 的补救口。
+    """
+    miss = [m for m in sorted(out.glob(f'*年/*月/*-{cls}.mdb')) if not m.with_suffix('.zip').exists()]
+    if not miss:
+        print(f'  没有缺 zip 的 {cls} 月库')
+        return 0
+    for m in miss:
+        z = m.with_suffix('.zip')
+        print(f'  + 补 zip {m.name} ({m.stat().st_size / 1e6:.1f} MB) …', end='', flush=True)
+        with zipfile.ZipFile(z, 'w', zipfile.ZIP_DEFLATED, compresslevel=6) as zf:
+            zf.write(m, m.name)
+        print(f' {z.stat().st_size / 1e6:.1f} MB')
+    print(f'  共补 {len(miss)} 个 zip')
+    return 0
+
+
 def main() -> int:
     ap = argparse.ArgumentParser(description='CSV → MDB(仿现场年度归档形态)+ 全量核对')
     ap.add_argument('--month', default=None, help='YYYY-MM(--verify 时可省)')
@@ -213,6 +245,7 @@ def main() -> int:
     ap.add_argument('--out', default=None)
     ap.add_argument('--limit-rows', type=int, default=0)
     ap.add_argument('--no-zip', action='store_true')
+    ap.add_argument('--rezip', action='store_true', help='给已在位但缺 zip 的月库补 zip(不重算)')
     ap.add_argument('--verify', action='store_true', help='全量核对报告(默认日历推算, 秒出)')
     ap.add_argument('--deep', action='store_true', help='--verify 深核对: 扫源 CSV 逐月计数')
     ap.add_argument('--refresh', action='store_true', help='--deep 时重扫源(不用缓存)')
@@ -220,6 +253,8 @@ def main() -> int:
     out = pathlib.Path(a.out) if a.out else (pathlib.Path(raw_station_dir()) / 'scada_mdb')
     if a.verify:
         return verify(out, a.cls, a.refresh, a.deep)
+    if a.rezip:
+        return rezip(out, a.cls)
     if not a.month:
         print('[X] 需要 --month(或用 --verify)')
         return 2