|
@@ -24,23 +24,55 @@ A2 定了四项数据源的落位: data/raw/<场站名称>/{scada_10min, 故障
|
|
|
(去掉"2025年油样"这一层, 直接落成 <台号>/<部件>/<pdf>,
|
|
(去掉"2025年油样"这一层, 直接落成 <台号>/<部件>/<pdf>,
|
|
|
与维护页说明"按台号/部件分目录"一致)
|
|
与维护页说明"按台号/部件分目录"一致)
|
|
|
|
|
|
|
|
-## 故意不做的事
|
|
|
|
|
|
|
+## 两组范围 (--scope)
|
|
|
|
|
|
|
|
- · 不落 `(3)现场检修记录/2025年检修记录` 与 `2026年检修记录`: 与 工作/风机故障记录/2025年故障记录、
|
|
|
|
|
- 2026年故障记录 **逐件同名同大小**(19/19 件), 是同一批月度汇总表的副本。落两遍会让"按年目录"
|
|
|
|
|
- 摄入看到两份重复台账 (2026-08-31 长停台账虚高 35 倍那类事故的同款成因: 快照重复必须归并, 不能叠加)。
|
|
|
|
|
- · 不落 `(8)…/振动分析报告/`: 属振动线 (CMS), 不在 A2 四项数据层内。
|
|
|
|
|
- · 不落 1分钟数据.zip / scada数据(如东)/ / 西门子4.0技术资料/ 等: A3 本轮范围是 A2 那四类。
|
|
|
|
|
|
|
+`--scope a2`(默认) 只落上面那四类, 即 A3 当时的范围, 语义不变。
|
|
|
|
|
+
|
|
|
|
|
+`--scope mech` 落的是**"能重算的缺口"**(2026-09-11 用户令: 缺什么从现场包里抽)。它解决的是
|
|
|
|
|
+"从零重算时, 某些产物因为源件不在 data/raw 而算不出来"这件事, 因此只挑**有消费者**的源:
|
|
|
|
|
+
|
|
|
|
|
+ <data/raw>/西门子4.0技术资料/ ← 如东风场数据.zip: 西门子4.0技术资料/** (317 件 2.8 GB)
|
|
|
|
|
+ + 如东风场数据/广核如海风电场西门子风机故障代码中英文对译表.xlsx
|
|
|
|
|
+ 依据: `src/ontology/kb_ingest.py` 里 `TECH = P.RAW_ROOT/'西门子4.0技术资料'`, 且 `rd()` 要读
|
|
|
|
|
+ `故障处理/故障处理手册.xlsx`、`广核如海风电场西门子风机故障代码中英文对译表.xlsx`、
|
|
|
|
|
+ `如海故障代码表.xlsx`、`维护相关/维护作业指导书.xlsx` 四个文件。前三个里**对译表在
|
|
|
|
|
+ 技术资料目录里没有**(包里它在 `scada数据(如东)/` 与包根各有一份), 所以单独补一条规则。
|
|
|
|
|
+ 效果: 本体层 `objects.json` 从"包内没有源件(⛔)"变成**可重算**。
|
|
|
|
|
+
|
|
|
|
|
+ <场站>/scada_1min/ ← 1分钟数据.zip 全部 38 个 <台号>.csv (01E.csv…37B.csv)
|
|
|
|
|
+ 依据: `src/windscada/config.py` 的 STATION_SUBDIRS 声明了这个子目录, `scan_stations.py`
|
|
|
|
|
+ 报它"缺"; 表头带 `source_file=如海风机1分钟数据/如海测点_2025-01.csv`, 与场配置
|
|
|
|
|
+ `src_farm_names=['如海','如东']` 同源(如海=如东项目)。**包内没有消费者**: 它补的是
|
|
|
|
|
+ 数据层完整性, 不改变任何页面数值 —— 落它是因为"扫描辨识"会把它报成缺口。
|
|
|
|
|
|
|
|
用法:
|
|
用法:
|
|
|
- python scripts/place_raw_data.py --dry-run # 只报要落什么, 不写盘
|
|
|
|
|
- python scripts/place_raw_data.py # 真落位 (已存在则覆盖)
|
|
|
|
|
|
|
+ python scripts/place_raw_data.py --dry-run # 只报要落什么, 不写盘 (默认 a2)
|
|
|
|
|
+ python scripts/place_raw_data.py # 真落位 (已存在则覆盖)
|
|
|
|
|
+ python scripts/place_raw_data.py --scope mech --dry-run # 看机理层/1min 会落什么
|
|
|
|
|
+ python scripts/place_raw_data.py --scope full # 两组一起
|
|
|
python scripts/place_raw_data.py --src D:\\别的现场数据目录
|
|
python scripts/place_raw_data.py --src D:\\别的现场数据目录
|
|
|
|
|
+
|
|
|
|
|
+## 故意不做的事 (两组范围共有的判断)
|
|
|
|
|
+
|
|
|
|
|
+ · 不落 `(3)现场检修记录/2025年检修记录` 与 `2026年检修记录`: 与 工作/风机故障记录/2025年故障记录、
|
|
|
|
|
+ 2026年故障记录 **逐件同名同大小**(19/19 件), 是同一批月度汇总表的副本。落两遍会让"按年目录"
|
|
|
|
|
+ 摄入看到两份重复台账 (2026-08-31 长停台账虚高 35 倍那类事故的同款成因: 快照重复必须归并, 不能叠加)。
|
|
|
|
|
+ · 不落 `(8)…/振动分析报告/`(12 份月度用印版 PDF): 是振动线 (CMS) 的**成品牌报告**, 不是可再加工的
|
|
|
|
|
+ 测量数据; 而 `windcms`/`m5_cms_tcm` 要的是 CMS 测点索引(handoff), 包里没有。
|
|
|
|
|
+ · 不落 `scada数据(如东)/**`(19 个月 × 12 个通道组 zip, 2.4 GB): 它是 `scada_10min/*.csv` 的**上游**
|
|
|
|
|
+ 原始通道导出, 包内没有任何脚本读它(构建器读的是已经平铺好的 10min CSV) —— 落了也不参与重算。
|
|
|
|
|
+ · 不落 `fastlog数据/`(4 件 WTG0x.xls): 全库搜 `fastlog` 只有 2 处**注释**提到它("运行态见证"),
|
|
|
|
|
+ 没有读取代码。
|
|
|
|
|
+ · 结论: 上面三项都是"有源件、无生成端/无消费者"。`genbearing_monthly`、`mblub_monthly`、
|
|
|
|
|
+ `yaw_dynamic_monthly`、`yaw1min_liveness`、`sector_power`、`duty_monthly`、`pc_monthly_bins`、
|
|
|
|
|
+ `thermal_monthly`、`structure.parquet`、`watch_channels_monthly` 这批产物同理 ——
|
|
|
|
|
+ 全库只有读取方、**0 处写入方**, 属于 v0.2.0 未附构建脚本(见 docs §4)。
|
|
|
"""
|
|
"""
|
|
|
from __future__ import annotations
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
import argparse
|
|
import argparse
|
|
|
import pathlib
|
|
import pathlib
|
|
|
|
|
+import os
|
|
|
import shutil
|
|
import shutil
|
|
|
import sys
|
|
import sys
|
|
|
import zipfile
|
|
import zipfile
|
|
@@ -53,15 +85,26 @@ DEFAULT_SRC = pathlib.Path(os.environ.get('GUANLAN_PLACE_SRC') or (ROOT / 'data'
|
|
|
ZIP_10MIN = '10分钟数据.zip'
|
|
ZIP_10MIN = '10分钟数据.zip'
|
|
|
ZIP_FARM = '如东风场数据.zip'
|
|
ZIP_FARM = '如东风场数据.zip'
|
|
|
|
|
|
|
|
-# (压缩包, 包内前缀, 目标子目录, 是否去掉前缀这一层)
|
|
|
|
|
|
|
+# (压缩包, 包内前缀, 目标, 是否去掉前缀这一层, 目标根: station= data/raw/<场站>/, raw= data/raw/)
|
|
|
RULES = [
|
|
RULES = [
|
|
|
- ('10分钟数据.zip', '', 'scada_10min', True),
|
|
|
|
|
- ('如东风场数据.zip', '如东风场数据/报警数据/', '故障报警', True),
|
|
|
|
|
- ('如东风场数据.zip', '如东风场数据/数据收集/更新/(2)故障记录(首发故障有标识)(2025.1-至今)/故障记录/', '故障报警', True),
|
|
|
|
|
- ('如东风场数据.zip', '如东风场数据/工作/风机故障记录/', '风机故障记录', True),
|
|
|
|
|
- ('如东风场数据.zip', '如东风场数据/数据收集/更新/(3)现场检修记录(2025.1-至今)/大部件维修记录.20240619143912557.xlsx', '风机故障记录', True),
|
|
|
|
|
- ('如东风场数据.zip', '如东风场数据/数据收集/更新/(8)风机振动数据、油液分析记录(2025.1-至今)/2025年油样/', '油样报告', True),
|
|
|
|
|
|
|
+ ('10分钟数据.zip', '', 'scada_10min', True, 'station'),
|
|
|
|
|
+ ('如东风场数据.zip', '如东风场数据/报警数据/', '故障报警', True, 'station'),
|
|
|
|
|
+ ('如东风场数据.zip', '如东风场数据/数据收集/更新/(2)故障记录(首发故障有标识)(2025.1-至今)/故障记录/', '故障报警', True, 'station'),
|
|
|
|
|
+ ('如东风场数据.zip', '如东风场数据/工作/风机故障记录/', '风机故障记录', True, 'station'),
|
|
|
|
|
+ ('如东风场数据.zip', '如东风场数据/数据收集/更新/(3)现场检修记录(2025.1-至今)/大部件维修记录.20240619143912557.xlsx', '风机故障记录', True, 'station'),
|
|
|
|
|
+ ('如东风场数据.zip', '如东风场数据/数据收集/更新/(8)风机振动数据、油液分析记录(2025.1-至今)/2025年油样/', '油样报告', True, 'station'),
|
|
|
]
|
|
]
|
|
|
|
|
+
|
|
|
|
|
+# 机理层与可选层 (--scope mech): 只挑**有消费者**的源 —— 目的是让"从零重算"能覆盖到本体层。
|
|
|
|
|
+RULES_MECH = [
|
|
|
|
|
+ # 本体层源件: kb_ingest.py 的 TECH = data/raw/西门子4.0技术资料 (317 件, 含故障处理/维护相关/各类图纸)
|
|
|
|
|
+ ('如东风场数据.zip', '如东风场数据/西门子4.0技术资料/', '西门子4.0技术资料', True, 'raw'),
|
|
|
|
|
+ # …但它要读的对译表不在技术资料目录里, 包里在别处 → 单独补一条 (见本文件开头"两组范围")
|
|
|
|
|
+ ('如东风场数据.zip', '如东风场数据/广核如海风电场西门子风机故障代码中英文对译表.xlsx', '西门子4.0技术资料', True, 'raw'),
|
|
|
|
|
+ # 数据层声明里的 scada_1min (config.STATION_SUBDIRS); 包内无消费者, 补的是数据层完整性
|
|
|
|
|
+ ('1分钟数据.zip', '', 'scada_1min', True, 'station'),
|
|
|
|
|
+]
|
|
|
|
|
+
|
|
|
# 说明性的"故意不落", 只在报告里列出来
|
|
# 说明性的"故意不落", 只在报告里列出来
|
|
|
SKIPPED = [
|
|
SKIPPED = [
|
|
|
('如东风场数据.zip', '如东风场数据/数据收集/更新/(3)现场检修记录(2025.1-至今)/2025年检修记录/',
|
|
('如东风场数据.zip', '如东风场数据/数据收集/更新/(3)现场检修记录(2025.1-至今)/2025年检修记录/',
|
|
@@ -69,7 +112,11 @@ SKIPPED = [
|
|
|
('如东风场数据.zip', '如东风场数据/数据收集/更新/(3)现场检修记录(2025.1-至今)/2026年检修记录/',
|
|
('如东风场数据.zip', '如东风场数据/数据收集/更新/(3)现场检修记录(2025.1-至今)/2026年检修记录/',
|
|
|
'与 工作/风机故障记录/2026年故障记录 逐件同名校验相同 (副本)'),
|
|
'与 工作/风机故障记录/2026年故障记录 逐件同名校验相同 (副本)'),
|
|
|
('如东风场数据.zip', '如东风场数据/数据收集/更新/(8)风机振动数据、油液分析记录(2025.1-至今)/振动分析报告/',
|
|
('如东风场数据.zip', '如东风场数据/数据收集/更新/(8)风机振动数据、油液分析记录(2025.1-至今)/振动分析报告/',
|
|
|
- '属振动线 (CMS), 不在 A2 四项数据层内'),
|
|
|
|
|
|
|
+ '振动线成品牌报告, 不是可再加工的测量数据; windcms 要的测点索引包里没有'),
|
|
|
|
|
+ ('如东风场数据.zip', '如东风场数据/scada数据(如东)/',
|
|
|
|
|
+ 'scada_10min/*.csv 的上游原始通道导出 (2.4 GB), 包内无脚本读它 → 不参与重算'),
|
|
|
|
|
+ ('如东风场数据.zip', '如东风场数据/fastlog数据/',
|
|
|
|
|
+ '全库搜 fastlog 只有 2 处注释提到, 无读取代码'),
|
|
|
]
|
|
]
|
|
|
|
|
|
|
|
|
|
|
|
@@ -93,13 +140,14 @@ def human(n: float) -> str:
|
|
|
return f'{n:.1f} GB'
|
|
return f'{n:.1f} GB'
|
|
|
|
|
|
|
|
|
|
|
|
|
-def plan(src: pathlib.Path, station: pathlib.Path):
|
|
|
|
|
- """→ [(zip, 目标文件, 包内条目名, 解压后大小)]; 只读压缩包目录, 不解压。"""
|
|
|
|
|
|
|
+def plan(src: pathlib.Path, station: pathlib.Path, rules):
|
|
|
|
|
+ """→ [(zip, 目标文件, 包内条目名, 解压后大小, 目标根, 目标名)]; 只读压缩包目录, 不解压。"""
|
|
|
out = []
|
|
out = []
|
|
|
- for zname, prefix, target, strip in RULES:
|
|
|
|
|
|
|
+ for zname, prefix, target, strip, root in rules:
|
|
|
zp = src / zname
|
|
zp = src / zname
|
|
|
if not zp.exists():
|
|
if not zp.exists():
|
|
|
raise SystemExit(f'缺压缩包: {zp}')
|
|
raise SystemExit(f'缺压缩包: {zp}')
|
|
|
|
|
+ base = station if root == 'station' else station.parent # 'raw' = data/raw (机理层不在场站目录下)
|
|
|
with zipfile.ZipFile(zp) as zf:
|
|
with zipfile.ZipFile(zp) as zf:
|
|
|
for info in zf.infolist():
|
|
for info in zf.infolist():
|
|
|
if info.is_dir():
|
|
if info.is_dir():
|
|
@@ -109,8 +157,14 @@ def plan(src: pathlib.Path, station: pathlib.Path):
|
|
|
continue
|
|
continue
|
|
|
rest = name[len(prefix):] if strip else name
|
|
rest = name[len(prefix):] if strip else name
|
|
|
if not rest:
|
|
if not rest:
|
|
|
- continue
|
|
|
|
|
- out.append((zname, station / target / rest, name, info.file_size))
|
|
|
|
|
|
|
+ # 前缀恰是条目本身 = 单文件规则 (如 '…/大部件维修记录.20240619143912557.xlsx')。
|
|
|
|
|
+ # 旧写法在这里直接 continue, 于是**单文件规则从来没落过东西** —— 2026-09-11 逮到两例:
|
|
|
|
|
+ # `大部件维修记录.20240619143912557.xlsx`(A2 规则, 从 A3 起就没落) 与
|
|
|
|
|
+ # `广核如海…对译表.xlsx`(mech 规则)。只有"目录条目"(prefix 以 / 结尾) 才该跳过。
|
|
|
|
|
+ if prefix.endswith('/'):
|
|
|
|
|
+ continue
|
|
|
|
|
+ rest = prefix.rstrip('/').rsplit('/', 1)[-1]
|
|
|
|
|
+ out.append((zname, base / target / rest, name, info.file_size, root, target))
|
|
|
return out
|
|
return out
|
|
|
|
|
|
|
|
|
|
|
|
@@ -118,6 +172,8 @@ def main() -> int:
|
|
|
ap = argparse.ArgumentParser()
|
|
ap = argparse.ArgumentParser()
|
|
|
ap.add_argument('--src', default=str(DEFAULT_SRC), help='现场数据目录 (默认 %(default)s)')
|
|
ap.add_argument('--src', default=str(DEFAULT_SRC), help='现场数据目录 (默认 %(default)s)')
|
|
|
ap.add_argument('--dry-run', action='store_true', help='只报计划, 不写盘')
|
|
ap.add_argument('--dry-run', action='store_true', help='只报计划, 不写盘')
|
|
|
|
|
+ ap.add_argument('--scope', choices=('a2', 'mech', 'full'), default='a2',
|
|
|
|
|
+ help='a2=只落四项数据层(A3 语义, 默认) · mech=机理层源件+1min · full=两组一起')
|
|
|
a = ap.parse_args()
|
|
a = ap.parse_args()
|
|
|
|
|
|
|
|
src = pathlib.Path(a.src)
|
|
src = pathlib.Path(a.src)
|
|
@@ -128,19 +184,19 @@ def main() -> int:
|
|
|
station = pathlib.Path(raw_station_dir())
|
|
station = pathlib.Path(raw_station_dir())
|
|
|
cfg = farm()
|
|
cfg = farm()
|
|
|
print(f'场站目录: {station} (来自 src/windscada/config.py raw_station={cfg.get("raw_station")!r})')
|
|
print(f'场站目录: {station} (来自 src/windscada/config.py raw_station={cfg.get("raw_station")!r})')
|
|
|
- print(f'现场数据: {src}\n')
|
|
|
|
|
|
|
+ print(f'现场数据: {src} 范围: --scope {a.scope}\n')
|
|
|
|
|
|
|
|
- items = plan(src, station)
|
|
|
|
|
|
|
+ rules = {'a2': RULES, 'mech': RULES_MECH, 'full': RULES + RULES_MECH}[a.scope]
|
|
|
|
|
+ items = plan(src, station, rules)
|
|
|
by_target = {}
|
|
by_target = {}
|
|
|
- for zname, dst, entry, size in items:
|
|
|
|
|
- t = dst.relative_to(station).parts[0]
|
|
|
|
|
- d = by_target.setdefault(t, [0, 0])
|
|
|
|
|
|
|
+ for zname, dst, entry, size, root, target in items:
|
|
|
|
|
+ d = by_target.setdefault((root, target), [0, 0])
|
|
|
d[0] += 1
|
|
d[0] += 1
|
|
|
d[1] += size
|
|
d[1] += size
|
|
|
- print('== 计划落位 ==')
|
|
|
|
|
- for t in ('scada_10min', '故障报警', '风机故障记录', '油样报告'):
|
|
|
|
|
- n, sz = by_target.get(t, (0, 0))
|
|
|
|
|
- print(f' {t:12s} {n:5d} 件 {human(sz):>10s} → {station / t}')
|
|
|
|
|
|
|
+ print(f'== 计划落位 (scope={a.scope}) ==')
|
|
|
|
|
+ for (root, target), (n, sz) in sorted(by_target.items(), key=lambda kv: kv[0][1]):
|
|
|
|
|
+ where = (station / target) if root == 'station' else (station.parent / target)
|
|
|
|
|
+ print(f' {target:18s} {n:5d} 件 {human(sz):>10s} → {where}')
|
|
|
print(f' 合计 {len(items)} 件, {human(sum(i[3] for i in items))}')
|
|
print(f' 合计 {len(items)} 件, {human(sum(i[3] for i in items))}')
|
|
|
|
|
|
|
|
print('\n== 故意不落 ==')
|
|
print('\n== 故意不落 ==')
|
|
@@ -153,7 +209,13 @@ def main() -> int:
|
|
|
|
|
|
|
|
print('\n== 落位 ==')
|
|
print('\n== 落位 ==')
|
|
|
done = 0
|
|
done = 0
|
|
|
- for zname, dst, entry, size in items:
|
|
|
|
|
|
|
+ skipped = 0
|
|
|
|
|
+ for zname, dst, entry, size, root, target in items:
|
|
|
|
|
+ # 已有同尺寸文件 = 已经是最新 → 跳过。重跑一次不该把 15 GB 原样再抄一遍
|
|
|
|
|
+ # (mech 范围 15.5 GB, a2 范围 14.7 GB; 2026-09-11 修单文件规则时就是靠这个避免整盘重写)。
|
|
|
|
|
+ if dst.exists() and dst.stat().st_size == size:
|
|
|
|
|
+ skipped += 1
|
|
|
|
|
+ continue
|
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
|
with zipfile.ZipFile(src / zname) as zf:
|
|
with zipfile.ZipFile(src / zname) as zf:
|
|
|
info = next(i for i in zf.infolist() if gbk_name(i).replace('\\', '/') == entry)
|
|
info = next(i for i in zf.infolist() if gbk_name(i).replace('\\', '/') == entry)
|
|
@@ -164,8 +226,8 @@ def main() -> int:
|
|
|
raise SystemExit(f'写出大小不符: {dst} 期望 {size} 实得 {got}')
|
|
raise SystemExit(f'写出大小不符: {dst} 期望 {size} 实得 {got}')
|
|
|
done += 1
|
|
done += 1
|
|
|
if done % 10 == 0 or size > 50 * 1024 * 1024:
|
|
if done % 10 == 0 or size > 50 * 1024 * 1024:
|
|
|
- print(f' [{done}/{len(items)}] {human(size):>10s} {dst.relative_to(station)}', flush=True)
|
|
|
|
|
- print(f'\n完成: {done} 件写入 {station}')
|
|
|
|
|
|
|
+ print(f' [{done}/{len(items)}] {human(size):>10s} {dst.relative_to(station.parent)}', flush=True)
|
|
|
|
|
+ print(f'\n完成 (scope={a.scope}): 新写/更新 {done} 件, 已是最新跳过 {skipped} 件, 共 {len(items)} 件')
|
|
|
return 0
|
|
return 0
|
|
|
|
|
|
|
|
|
|
|