| 1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162 |
- # -*- coding: utf-8 -*-
- """§3.4 10min 真时均派生层 (秒级/sub-10min cleaned → 10min, 非瞬时快照).
- per signal: mean/std/min/max/count(有效样本) + 可选 power-p90 / vibration-max / 状态码 fractions。
- 抽自 scripts/zyx_sec_aggregate.py 已跑逻辑 (floor("10min") + per-bin agg)。Batch2 ON-2 依赖之一。
- 纯函数, 无 I/O; 下游(性能/偏航复核/桨距交叉验证)吃此层。
- """
- from __future__ import annotations
- import pandas as pd
- def derive_10min(df: pd.DataFrame, *, time_col: str, signal_cols, machine_col: str | None = None,
- freq: str = "10min", power_col: str | None = None, vib_cols=(),
- status_col: str | None = None, code_sets: dict | None = None,
- min_valid: int = 30) -> pd.DataFrame:
- """秒级/sub-10min cleaned df → 10min 真时均层。
- 参数:
- time_col 时间列 (可解析为 datetime)
- signal_cols 要派生 mean/std/min/max/count 的模拟量列
- machine_col 机号列 (None = 单机/已筛)
- power_col 有功功率列 → 额外出 _p90 (满发包络)
- vib_cols 振动列 → 额外出 _max (峰值, §3.2 不混池)
- status_col + code_sets 状态码列 + {名: 码集} → 出 <名>_frac (gen/limit/curt/stop)
- min_valid 每窗有效样本下限; 不足 → valid=False (下游决定是否弃, 不在此清)
- 返回 long df: [machine?, bin, n, valid, <sig>_{mean,std,min,max,n}, (power_p90), (vib_max), (<名>_frac)]
- """
- d = df.copy()
- d["_t"] = pd.to_datetime(d[time_col], errors="coerce")
- d = d.dropna(subset=["_t"])
- if d.empty:
- raise ValueError(f"derive_10min: time_col={time_col!r} 解析后无有效行")
- d["_bin"] = d["_t"].dt.floor(freq)
- gkeys = ([machine_col] if machine_col else []) + ["_bin"]
- sig = [c for c in signal_cols if c in d.columns]
- for c in sig:
- d[c] = pd.to_numeric(d[c], errors="coerce")
- grp = d.groupby(gkeys, sort=True)
- out = pd.DataFrame({"n": grp.size()})
- for c in sig:
- gc = grp[c]
- out[f"{c}_mean"] = gc.mean()
- out[f"{c}_std"] = gc.std()
- out[f"{c}_min"] = gc.min()
- out[f"{c}_max"] = gc.max()
- out[f"{c}_n"] = gc.count() # 有效(非NaN)样本数 — 真时均的可信度
- if power_col and power_col in d.columns:
- out[f"{power_col}_p90"] = grp[power_col].apply(lambda s: pd.to_numeric(s, errors="coerce").quantile(0.9))
- for vc in vib_cols:
- if vc in d.columns:
- out[f"{vc}_max"] = grp[vc].apply(lambda s: pd.to_numeric(s, errors="coerce").max())
- if status_col and status_col in d.columns and code_sets:
- st = pd.to_numeric(d[status_col], errors="coerce")
- g2 = d.assign(_st=st).groupby(gkeys)["_st"]
- for name, codes in code_sets.items():
- cs = set(codes)
- out[f"{name}_frac"] = g2.apply(lambda s, c=cs: s.isin(c).mean())
- out["valid"] = out["n"] >= min_valid
- return out.reset_index().rename(columns={"_bin": "bin"})
|