derive10min.py 3.0 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162
  1. # -*- coding: utf-8 -*-
  2. """§3.4 10min 真时均派生层 (秒级/sub-10min cleaned → 10min, 非瞬时快照).
  3. per signal: mean/std/min/max/count(有效样本) + 可选 power-p90 / vibration-max / 状态码 fractions。
  4. 抽自 scripts/zyx_sec_aggregate.py 已跑逻辑 (floor("10min") + per-bin agg)。Batch2 ON-2 依赖之一。
  5. 纯函数, 无 I/O; 下游(性能/偏航复核/桨距交叉验证)吃此层。
  6. """
  7. from __future__ import annotations
  8. import pandas as pd
  9. def derive_10min(df: pd.DataFrame, *, time_col: str, signal_cols, machine_col: str | None = None,
  10. freq: str = "10min", power_col: str | None = None, vib_cols=(),
  11. status_col: str | None = None, code_sets: dict | None = None,
  12. min_valid: int = 30) -> pd.DataFrame:
  13. """秒级/sub-10min cleaned df → 10min 真时均层。
  14. 参数:
  15. time_col 时间列 (可解析为 datetime)
  16. signal_cols 要派生 mean/std/min/max/count 的模拟量列
  17. machine_col 机号列 (None = 单机/已筛)
  18. power_col 有功功率列 → 额外出 _p90 (满发包络)
  19. vib_cols 振动列 → 额外出 _max (峰值, §3.2 不混池)
  20. status_col + code_sets 状态码列 + {名: 码集} → 出 <名>_frac (gen/limit/curt/stop)
  21. min_valid 每窗有效样本下限; 不足 → valid=False (下游决定是否弃, 不在此清)
  22. 返回 long df: [machine?, bin, n, valid, <sig>_{mean,std,min,max,n}, (power_p90), (vib_max), (<名>_frac)]
  23. """
  24. d = df.copy()
  25. d["_t"] = pd.to_datetime(d[time_col], errors="coerce")
  26. d = d.dropna(subset=["_t"])
  27. if d.empty:
  28. raise ValueError(f"derive_10min: time_col={time_col!r} 解析后无有效行")
  29. d["_bin"] = d["_t"].dt.floor(freq)
  30. gkeys = ([machine_col] if machine_col else []) + ["_bin"]
  31. sig = [c for c in signal_cols if c in d.columns]
  32. for c in sig:
  33. d[c] = pd.to_numeric(d[c], errors="coerce")
  34. grp = d.groupby(gkeys, sort=True)
  35. out = pd.DataFrame({"n": grp.size()})
  36. for c in sig:
  37. gc = grp[c]
  38. out[f"{c}_mean"] = gc.mean()
  39. out[f"{c}_std"] = gc.std()
  40. out[f"{c}_min"] = gc.min()
  41. out[f"{c}_max"] = gc.max()
  42. out[f"{c}_n"] = gc.count() # 有效(非NaN)样本数 — 真时均的可信度
  43. if power_col and power_col in d.columns:
  44. out[f"{power_col}_p90"] = grp[power_col].apply(lambda s: pd.to_numeric(s, errors="coerce").quantile(0.9))
  45. for vc in vib_cols:
  46. if vc in d.columns:
  47. out[f"{vc}_max"] = grp[vc].apply(lambda s: pd.to_numeric(s, errors="coerce").max())
  48. if status_col and status_col in d.columns and code_sets:
  49. st = pd.to_numeric(d[status_col], errors="coerce")
  50. g2 = d.assign(_st=st).groupby(gkeys)["_st"]
  51. for name, codes in code_sets.items():
  52. cs = set(codes)
  53. out[f"{name}_frac"] = g2.apply(lambda s, c=cs: s.isin(c).mean())
  54. out["valid"] = out["n"] >= min_valid
  55. return out.reset_index().rename(columns={"_bin": "bin"})