| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286 |
- #!/usr/bin/env python3
- # -*- coding: utf-8 -*-
- r"""P12-A:Java 后端迁移的**裁判**——契约端点逐一比对 Java 与 Python 的响应。
- 用途(与前端对拍门同思路:门禁当裁判,逐条收敛):
- · 读冻契约 `app_backEnd/.../contract/http_api_v1.json`(42 条);
- · 对每条端点,分别请求 Python 侧与 Java 侧,做**深度相等**比较(JSON 归一化后);
- · 输出计分板:`已一致 / 有差异 / Java 未实现`,并把"必须一致"的端点清单交给 `--require` 用于门禁。
- 用法:
- python scripts/java_contract_parity.py [--gateway URL] [--python-base /detail] [--java-base /java]
- [--require kpi,fleet,...] [--limit N]
- [--java-kinds detail_api]
- ★预热:比对前两侧各 GET 一次(丢弃结果)再取第二次 —— 带"按窗重算/进程内缓存"的端点(如 /api/fleet)
- 冷启时点不同会报假差异(实逮 13≠15,预热后均 15)。
- """
- from __future__ import annotations
- import argparse
- import json
- import pathlib
- import sys
- import urllib.error
- import urllib.request
- ROOT = pathlib.Path(__file__).resolve().parents[1]
- CONTRACT = ROOT / 'app_backEnd/app_backEnd_guanlan/contract/http_api_v1.json'
- def get(url: str, timeout: float = 60.0):
- req = urllib.request.Request(url, headers={'Accept': 'application/json'})
- try:
- with urllib.request.urlopen(req, timeout=timeout) as r:
- body = r.read().decode('utf-8', 'replace')
- try:
- return r.status, json.loads(body)
- except Exception:
- return r.status, body[:400]
- except urllib.error.HTTPError as e:
- return e.code, (e.read().decode('utf-8', 'replace')[:200])
- except Exception as e: # noqa: BLE001
- return 0, f'{type(e).__name__}: {e}'
- def get_bytes(url: str, timeout: float = 900.0):
- """取原始字节(二进制端点用);失败返回 None。"""
- import urllib.error
- import urllib.request
- try:
- with urllib.request.urlopen(urllib.request.Request(url), timeout=timeout) as r:
- return r.read()
- except Exception: # noqa: BLE001
- return None
- def post_json(url: str, payload: dict, timeout: float = 300.0):
- """POST JSON(裁判的 POST 能力):返回 (status, 解析后的 JSON 或错误文本)。"""
- import urllib.error
- import urllib.request
- data = json.dumps(payload).encode('utf-8')
- req = urllib.request.Request(url, data=data, headers={'Content-Type': 'application/json'})
- try:
- with urllib.request.urlopen(req, timeout=timeout) as r:
- body = r.read().decode('utf-8', 'replace')
- try:
- return r.status, json.loads(body)
- except Exception: # noqa: BLE001
- return r.status, body[:300]
- except urllib.error.HTTPError as e:
- return e.code, e.read().decode('utf-8', 'replace')[:300]
- except Exception as e: # noqa: BLE001
- return 0, '%s: %s' % (type(e).__name__, e)
- def norm(x):
- """比较用归一化:字典按键排序;列表保序(顺序也是行为)。"""
- if isinstance(x, dict):
- return {k: norm(v) for k, v in sorted(x.items())}
- if isinstance(x, list):
- return [norm(v) for v in x]
- return x
- def doc_text(b: bytes) -> str | None:
- """Office(zip) 文档 → 归一化文本:剔除时间戳/文档属性后逐条目拼接(用于二进制端点的等价判定)。"""
- if not b[:2] == b'PK':
- return None
- import io
- import re as _re
- import zipfile
- out = []
- with zipfile.ZipFile(io.BytesIO(b)) as z:
- for name in sorted(z.namelist()):
- if name.endswith('.xml') or name.endswith('.rels'):
- s = z.read(name).decode('utf-8', 'replace')
- s = _re.sub(r'<dcterms:(created|modified)[^>]*>[^<]*</dcterms:\1>', '', s)
- s = _re.sub(r'<cp:lastModifiedBy>[^<]*</cp:lastModifiedBy>', '', s)
- s = _re.sub(r'<TotalTime>[^<]*</TotalTime>', '', s)
- out.append(name + '\n' + s)
- return '\n'.join(out)
- def first_diff(a, b, path='$', depth=0):
- if depth > 6:
- return None
- if type(a) is not type(b) and not (isinstance(a, (int, float)) and isinstance(b, (int, float))):
- return f'{path}: 类型 {type(a).__name__} ≠ {type(b).__name__}'
- if isinstance(a, dict):
- for k in sorted(set(a) | set(b)):
- if k not in a:
- return f'{path}.{k}: Java 多键'
- if k not in b:
- return f'{path}.{k}: Java 缺键'
- d = first_diff(a[k], b[k], f'{path}.{k}', depth + 1)
- if d:
- return d
- return None
- if isinstance(a, list):
- if len(a) != len(b):
- return f'{path}: 长度 {len(a)} ≠ {len(b)}'
- for i, (x, y) in enumerate(zip(a, b)):
- d = first_diff(x, y, f'{path}[{i}]', depth + 1)
- if d:
- return d
- return None
- if a != b:
- sa, sb = str(a), str(b)
- return f'{path}: {sa[:60]!r} ≠ {sb[:60]!r}'
- return None
- # 登记忽略字段(每条都要写理由;只忽略"有状态"字段,不忽略实现口径)。
- #
- # `/api/reload` 的 `old_stamp` = 清缓存**前**的产物指纹 —— 取决于比对顺序与两侧各自的缓存状态:
- # 先被问的一侧缓存里有指纹(返回字符串),后被问的一侧可能已被其它探测清空(None)。
- # 其余字段(ok / files / stamp)照常严格比对。
- IGNORE_PATHS: dict[str, tuple[str, str]] = {
- '/api/reload': ('$.old_stamp', '清缓存前的指纹:取决于比对顺序与各自的缓存状态,非实现差异'),
- }
- def main() -> int:
- ap = argparse.ArgumentParser()
- ap.add_argument('--gateway', default='http://127.0.0.1:28084')
- ap.add_argument('--python-base', default='/detail')
- ap.add_argument('--java-base', default='/java')
- ap.add_argument('--require', default='', help='逗号分隔:这些端点必须一致,否则退出码 1')
- ap.add_argument('--limit', type=int, default=0)
- ap.add_argument('--java-kinds', default='detail_api',
- help='哪些 kind 属于 Java 后端迁移范围(默认 detail_api;逗号分隔)')
- a = ap.parse_args()
- contract = json.loads(CONTRACT.read_text(encoding='utf-8'))
- eps = contract['endpoints']
- rows = eps if isinstance(eps, list) else [dict(path=k, **(v if isinstance(v, dict) else {})) for k, v in eps.items()]
- if a.limit:
- rows = rows[:a.limit]
- java_kinds = {x.strip() for x in a.java_kinds.split(',') if x.strip()}
- same, diff, missing, skipped, other, unsettled, ignored = [], [], [], [], [], [], []
- scope_total = 0
- for e in rows:
- path = str(e.get('path') or e.get('url'))
- method = str(e.get('method') or 'GET').upper()
- kind = str(e.get('kind') or '?')
- if kind not in java_kinds:
- other.append((path, kind))
- continue
- scope_total += 1
- # ★ 一律真的问两侧:Java 走 /java 前缀、Python 走 /detail 前缀(灰度期两个并存入口)
- jurl = a.gateway.rstrip('/') + a.java_base + path
- purl = a.gateway.rstrip('/') + a.python_base + path
- if method != 'GET':
- # ★POST 能力(P12):按"契约抓取时的最小请求"发空体 {} —— 两侧都应回同样的错误语义
- # (如 /api/ask 空问题 ⇒ {err: 问题为空})。请求体若将来需要参数,再按契约补。
- js, jb = post_json(jurl, {})
- ps, pb = post_json(purl, {})
- if js in (0, 404):
- missing.append((path, 'Java HTTP %s' % js))
- continue
- d = first_diff(norm(jb), norm(pb))
- if d:
- diff.append((path, d))
- else:
- same.append(path)
- continue
- # ★二进制端点(契约 ctype 非 JSON,如 /api/rpt_export 的 docx):走"字节 + 解包文本"比对
- ctype = str(e.get('ctype') or '')
- if ctype and 'json' not in ctype.lower():
- jb_b = get_bytes(jurl)
- pb_b = get_bytes(purl)
- if jb_b is None or pb_b is None:
- diff.append((path, 'Java 或 Python 未返回字节(HTTP %s/%s)' % (jb_b, pb_b)))
- continue
- tj, tp = doc_text(jb_b), doc_text(pb_b)
- if tj is not None and tp is not None:
- if tj == tp:
- same.append(path)
- else:
- d = first_diff({'t': tj}, {'t': tp})
- diff.append((path, '解包文本不一致:' + str(d)[:120]))
- else:
- diff.append((path, '非 zip 字节:Java %d B / Python %d B' % (len(jb_b), len(pb_b))))
- continue
- # ★预热:该端点可能带"按窗重算 + 进程内缓存"(如 /api/fleet、/api/curves),
- # 冷启时点不同会报假差异。做法:各取一次丢弃结果 → 若仍有 *_pending=true 就轮询到落定 → 再比第二次。
- get(jurl)
- get(purl)
- # 再各取一次:对 /api/reload 这类**清缓存副作用**端点,连比两次会让"先比的一侧"持有旧指纹、
- # 后比的一侧为 None(实逮 old_stamp str ≠ None)⇒ 两次预热后两侧都处于"刚清过"的稳态。
- get(jurl)
- get(purl)
- def pending(x):
- if isinstance(x, dict):
- if x.get('building') is True: # 曲线等服务端后台算时的"建设中"标记
- return True
- return any(str(k).endswith('_pending') and v is True for k, v in x.items()) \
- or any(pending(v) for v in x.values())
- if isinstance(x, list):
- return any(pending(v) for v in x)
- return False
- settled = True
- for _ in range(20):
- js, jb = get(jurl)
- ps, pb = get(purl)
- if not pending(jb) and not pending(pb):
- settled = True
- break
- settled = False
- import time as _t
- _t.sleep(15)
- if not settled:
- unsettled.append((path, '两侧未在等待上限内落定(按窗重算仍在跑)—— 时点问题,不计为差异'))
- continue
- # ★只按 HTTP 状态判定(曾按正文里是否有 "404" 判 ⇒ 把"已实现但正文含该字样"的端点误判为未实现)
- if js in (0, 404):
- missing.append((path, f'Java HTTP {js}'))
- continue
- if js != 200 or ps != 200:
- diff.append((path, f'HTTP Java {js} / Python {ps}'))
- continue
- d = first_diff(norm(jb), norm(pb))
- if d and path in IGNORE_PATHS and d.startswith(IGNORE_PATHS[path][0]):
- ignored.append((path, IGNORE_PATHS[path][0], IGNORE_PATHS[path][1]))
- same.append(path)
- elif d:
- diff.append((path, d))
- else:
- same.append(path)
- print('Java ↔ Python 契约端点比对(裁判:冻契约 %d 条;Java 范围 kind=%s 共 %d 条)'
- % (len(rows), ','.join(sorted(java_kinds)), scope_total))
- print(' 一致 %d/%d · 有差异 %d · 未实现 %d · 跳过 %d · 未落定 %d(范围外 %d 条:%s)\n'
- % (len(same), scope_total, len(diff), len(missing), len(skipped), len(unsettled), len(other),
- ', '.join(sorted({k for _, k in other})) or '-'))
- if same:
- print(' [一致] ' + ', '.join(same[:12]) + (' …' if len(same) > 12 else ''))
- for p, why in diff[:10]:
- print(' [差异] %-28s %s' % (p, why))
- for p, why in missing[:10]:
- print(' [未实现] %-26s %s' % (p, why))
- for p, why in skipped[:6]:
- print(' [跳过] %-28s %s' % (p, why))
- for p, why in unsettled[:6]:
- print(' [未落定] %-26s %s' % (p, why))
- for p, f, why in ignored[:6]:
- print(' [登记忽略] %-24s %s —— %s' % (p, f, why))
- need = [x.strip() for x in a.require.split(',') if x.strip()]
- bad = [p for p in need if p not in same]
- if need:
- print('\n门禁要求一致 %d 条 → %s' % (len(need), '通过' if not bad else '未通过: ' + ', '.join(bad)))
- print('\n结论: %s' % ('Java 范围端点全部一致' if not diff and not missing else
- '迁移进行中(%d/%d 已一致)' % (len(same), scope_total)))
- return 0 if not bad else 1
- if __name__ == '__main__':
- raise SystemExit(main())
|