java_contract_parity.py 11 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. r"""P12-A:Java 后端迁移的**裁判**——契约端点逐一比对 Java 与 Python 的响应。
  4. 用途(与前端对拍门同思路:门禁当裁判,逐条收敛):
  5. · 读冻契约 `app_backEnd/.../contract/http_api_v1.json`(42 条);
  6. · 对每条端点,分别请求 Python 侧与 Java 侧,做**深度相等**比较(JSON 归一化后);
  7. · 输出计分板:`已一致 / 有差异 / Java 未实现`,并把"必须一致"的端点清单交给 `--require` 用于门禁。
  8. 用法:
  9. python scripts/java_contract_parity.py [--gateway URL] [--python-base /detail] [--java-base /java]
  10. [--require kpi,fleet,...] [--limit N]
  11. [--java-kinds detail_api]
  12. ★预热:比对前两侧各 GET 一次(丢弃结果)再取第二次 —— 带"按窗重算/进程内缓存"的端点(如 /api/fleet)
  13. 冷启时点不同会报假差异(实逮 13≠15,预热后均 15)。
  14. """
  15. from __future__ import annotations
  16. import argparse
  17. import json
  18. import pathlib
  19. import sys
  20. import urllib.error
  21. import urllib.request
  22. ROOT = pathlib.Path(__file__).resolve().parents[1]
  23. CONTRACT = ROOT / 'app_backEnd/app_backEnd_guanlan/contract/http_api_v1.json'
  24. def get(url: str, timeout: float = 60.0):
  25. req = urllib.request.Request(url, headers={'Accept': 'application/json'})
  26. try:
  27. with urllib.request.urlopen(req, timeout=timeout) as r:
  28. body = r.read().decode('utf-8', 'replace')
  29. try:
  30. return r.status, json.loads(body)
  31. except Exception:
  32. return r.status, body[:400]
  33. except urllib.error.HTTPError as e:
  34. return e.code, (e.read().decode('utf-8', 'replace')[:200])
  35. except Exception as e: # noqa: BLE001
  36. return 0, f'{type(e).__name__}: {e}'
  37. def get_bytes(url: str, timeout: float = 900.0):
  38. """取原始字节(二进制端点用);失败返回 None。"""
  39. import urllib.error
  40. import urllib.request
  41. try:
  42. with urllib.request.urlopen(urllib.request.Request(url), timeout=timeout) as r:
  43. return r.read()
  44. except Exception: # noqa: BLE001
  45. return None
  46. def norm(x):
  47. """比较用归一化:字典按键排序;列表保序(顺序也是行为)。"""
  48. if isinstance(x, dict):
  49. return {k: norm(v) for k, v in sorted(x.items())}
  50. if isinstance(x, list):
  51. return [norm(v) for v in x]
  52. return x
  53. def doc_text(b: bytes) -> str | None:
  54. """Office(zip) 文档 → 归一化文本:剔除时间戳/文档属性后逐条目拼接(用于二进制端点的等价判定)。"""
  55. if not b[:2] == b'PK':
  56. return None
  57. import io
  58. import re as _re
  59. import zipfile
  60. out = []
  61. with zipfile.ZipFile(io.BytesIO(b)) as z:
  62. for name in sorted(z.namelist()):
  63. if name.endswith('.xml') or name.endswith('.rels'):
  64. s = z.read(name).decode('utf-8', 'replace')
  65. s = _re.sub(r'<dcterms:(created|modified)[^>]*>[^<]*</dcterms:\1>', '', s)
  66. s = _re.sub(r'<cp:lastModifiedBy>[^<]*</cp:lastModifiedBy>', '', s)
  67. s = _re.sub(r'<TotalTime>[^<]*</TotalTime>', '', s)
  68. out.append(name + '\n' + s)
  69. return '\n'.join(out)
  70. def first_diff(a, b, path='$', depth=0):
  71. if depth > 6:
  72. return None
  73. if type(a) is not type(b) and not (isinstance(a, (int, float)) and isinstance(b, (int, float))):
  74. return f'{path}: 类型 {type(a).__name__} ≠ {type(b).__name__}'
  75. if isinstance(a, dict):
  76. for k in sorted(set(a) | set(b)):
  77. if k not in a:
  78. return f'{path}.{k}: Java 多键'
  79. if k not in b:
  80. return f'{path}.{k}: Java 缺键'
  81. d = first_diff(a[k], b[k], f'{path}.{k}', depth + 1)
  82. if d:
  83. return d
  84. return None
  85. if isinstance(a, list):
  86. if len(a) != len(b):
  87. return f'{path}: 长度 {len(a)} ≠ {len(b)}'
  88. for i, (x, y) in enumerate(zip(a, b)):
  89. d = first_diff(x, y, f'{path}[{i}]', depth + 1)
  90. if d:
  91. return d
  92. return None
  93. if a != b:
  94. sa, sb = str(a), str(b)
  95. return f'{path}: {sa[:60]!r} ≠ {sb[:60]!r}'
  96. return None
  97. # 登记忽略字段(每条都要写理由;只忽略"有状态"字段,不忽略实现口径)。
  98. #
  99. # `/api/reload` 的 `old_stamp` = 清缓存**前**的产物指纹 —— 取决于比对顺序与两侧各自的缓存状态:
  100. # 先被问的一侧缓存里有指纹(返回字符串),后被问的一侧可能已被其它探测清空(None)。
  101. # 其余字段(ok / files / stamp)照常严格比对。
  102. IGNORE_PATHS: dict[str, tuple[str, str]] = {
  103. '/api/reload': ('$.old_stamp', '清缓存前的指纹:取决于比对顺序与各自的缓存状态,非实现差异'),
  104. }
  105. def main() -> int:
  106. ap = argparse.ArgumentParser()
  107. ap.add_argument('--gateway', default='http://127.0.0.1:28084')
  108. ap.add_argument('--python-base', default='/detail')
  109. ap.add_argument('--java-base', default='/java')
  110. ap.add_argument('--require', default='', help='逗号分隔:这些端点必须一致,否则退出码 1')
  111. ap.add_argument('--limit', type=int, default=0)
  112. ap.add_argument('--java-kinds', default='detail_api',
  113. help='哪些 kind 属于 Java 后端迁移范围(默认 detail_api;逗号分隔)')
  114. a = ap.parse_args()
  115. contract = json.loads(CONTRACT.read_text(encoding='utf-8'))
  116. eps = contract['endpoints']
  117. rows = eps if isinstance(eps, list) else [dict(path=k, **(v if isinstance(v, dict) else {})) for k, v in eps.items()]
  118. if a.limit:
  119. rows = rows[:a.limit]
  120. java_kinds = {x.strip() for x in a.java_kinds.split(',') if x.strip()}
  121. same, diff, missing, skipped, other, unsettled, ignored = [], [], [], [], [], [], []
  122. scope_total = 0
  123. for e in rows:
  124. path = str(e.get('path') or e.get('url'))
  125. method = str(e.get('method') or 'GET').upper()
  126. kind = str(e.get('kind') or '?')
  127. if kind not in java_kinds:
  128. other.append((path, kind))
  129. continue
  130. scope_total += 1
  131. if method != 'GET':
  132. skipped.append((path, 'POST 类端点本轮不自动比(需构造请求体)'))
  133. continue
  134. # ★ 一律真的问两侧:Java 走 /java 前缀、Python 走 /detail 前缀(灰度期两个并存入口)
  135. jurl = a.gateway.rstrip('/') + a.java_base + path
  136. purl = a.gateway.rstrip('/') + a.python_base + path
  137. # ★二进制端点(契约 ctype 非 JSON,如 /api/rpt_export 的 docx):走"字节 + 解包文本"比对
  138. ctype = str(e.get('ctype') or '')
  139. if ctype and 'json' not in ctype.lower():
  140. jb_b = get_bytes(jurl)
  141. pb_b = get_bytes(purl)
  142. if jb_b is None or pb_b is None:
  143. diff.append((path, 'Java 或 Python 未返回字节(HTTP %s/%s)' % (jb_b, pb_b)))
  144. continue
  145. tj, tp = doc_text(jb_b), doc_text(pb_b)
  146. if tj is not None and tp is not None:
  147. if tj == tp:
  148. same.append(path)
  149. else:
  150. d = first_diff({'t': tj}, {'t': tp})
  151. diff.append((path, '解包文本不一致:' + str(d)[:120]))
  152. else:
  153. diff.append((path, '非 zip 字节:Java %d B / Python %d B' % (len(jb_b), len(pb_b))))
  154. continue
  155. # ★预热:该端点可能带"按窗重算 + 进程内缓存"(如 /api/fleet、/api/curves),
  156. # 冷启时点不同会报假差异。做法:各取一次丢弃结果 → 若仍有 *_pending=true 就轮询到落定 → 再比第二次。
  157. get(jurl)
  158. get(purl)
  159. # 再各取一次:对 /api/reload 这类**清缓存副作用**端点,连比两次会让"先比的一侧"持有旧指纹、
  160. # 后比的一侧为 None(实逮 old_stamp str ≠ None)⇒ 两次预热后两侧都处于"刚清过"的稳态。
  161. get(jurl)
  162. get(purl)
  163. def pending(x):
  164. if isinstance(x, dict):
  165. if x.get('building') is True: # 曲线等服务端后台算时的"建设中"标记
  166. return True
  167. return any(str(k).endswith('_pending') and v is True for k, v in x.items()) \
  168. or any(pending(v) for v in x.values())
  169. if isinstance(x, list):
  170. return any(pending(v) for v in x)
  171. return False
  172. settled = True
  173. for _ in range(20):
  174. js, jb = get(jurl)
  175. ps, pb = get(purl)
  176. if not pending(jb) and not pending(pb):
  177. settled = True
  178. break
  179. settled = False
  180. import time as _t
  181. _t.sleep(15)
  182. if not settled:
  183. unsettled.append((path, '两侧未在等待上限内落定(按窗重算仍在跑)—— 时点问题,不计为差异'))
  184. continue
  185. # ★只按 HTTP 状态判定(曾按正文里是否有 "404" 判 ⇒ 把"已实现但正文含该字样"的端点误判为未实现)
  186. if js in (0, 404):
  187. missing.append((path, f'Java HTTP {js}'))
  188. continue
  189. if js != 200 or ps != 200:
  190. diff.append((path, f'HTTP Java {js} / Python {ps}'))
  191. continue
  192. d = first_diff(norm(jb), norm(pb))
  193. if d and path in IGNORE_PATHS and d.startswith(IGNORE_PATHS[path][0]):
  194. ignored.append((path, IGNORE_PATHS[path][0], IGNORE_PATHS[path][1]))
  195. same.append(path)
  196. elif d:
  197. diff.append((path, d))
  198. else:
  199. same.append(path)
  200. print('Java ↔ Python 契约端点比对(裁判:冻契约 %d 条;Java 范围 kind=%s 共 %d 条)'
  201. % (len(rows), ','.join(sorted(java_kinds)), scope_total))
  202. print(' 一致 %d/%d · 有差异 %d · 未实现 %d · 跳过 %d · 未落定 %d(范围外 %d 条:%s)\n'
  203. % (len(same), scope_total, len(diff), len(missing), len(skipped), len(unsettled), len(other),
  204. ', '.join(sorted({k for _, k in other})) or '-'))
  205. if same:
  206. print(' [一致] ' + ', '.join(same[:12]) + (' …' if len(same) > 12 else ''))
  207. for p, why in diff[:10]:
  208. print(' [差异] %-28s %s' % (p, why))
  209. for p, why in missing[:10]:
  210. print(' [未实现] %-26s %s' % (p, why))
  211. for p, why in skipped[:6]:
  212. print(' [跳过] %-28s %s' % (p, why))
  213. for p, why in unsettled[:6]:
  214. print(' [未落定] %-26s %s' % (p, why))
  215. for p, f, why in ignored[:6]:
  216. print(' [登记忽略] %-24s %s —— %s' % (p, f, why))
  217. need = [x.strip() for x in a.require.split(',') if x.strip()]
  218. bad = [p for p in need if p not in same]
  219. if need:
  220. print('\n门禁要求一致 %d 条 → %s' % (len(need), '通过' if not bad else '未通过: ' + ', '.join(bad)))
  221. print('\n结论: %s' % ('Java 范围端点全部一致' if not diff and not missing else
  222. '迁移进行中(%d/%d 已一致)' % (len(same), scope_total)))
  223. return 0 if not bad else 1
  224. if __name__ == '__main__':
  225. raise SystemExit(main())