"""Deterministic SYNTHETIC dataset for the M2 eval baseline. There is no historical Hubei data in this repository yet (docs/12 §2 历史重放集 comes from the event log once the system runs; partner data is OPEN-QUESTION D-class). Until real data lands, the eval harness needs *something* with realistic structure — daily load shape, PV bell with cloud days, price with an evening peak and occasional spikes — so metrics, calibration tests and the reproducibility check exercise real code paths. Every number here is invented for shape only. Baselines computed on it are baselines of the *harness*, not of the business KPI (预测误差 ≤ 8% is judged on real data, docs/12 §4). The dataset file says so in its ``meta``. """ from __future__ import annotations import argparse import json from datetime import date, timedelta from pathlib import Path import numpy as np from .numeric import INTERVALS, SCALE_MW, SCALE_PRICE, quantize GENERATOR = "vpp_skills.synthetic" GENERATOR_VERSION = "0.1.0" def _profiles() -> tuple[np.ndarray, np.ndarray, np.ndarray]: t = np.arange(INTERVALS) / 4.0 # hours load = 0.55 + 0.25 * np.exp(-((t - 11) ** 2) / 8) + 0.35 * np.exp(-((t - 19.5) ** 2) / 6) - 0.15 * np.exp(-((t - 3) ** 2) / 10) pv = np.clip(np.sin(np.pi * (t - 6.5) / 12.5), 0.0, None) ** 1.4 pv[(t < 6.5) | (t > 19.0)] = 0.0 price = 0.7 + 0.2 * np.exp(-((t - 11) ** 2) / 6) + 0.5 * np.exp(-((t - 19) ** 2) / 4) - 0.25 * np.exp(-((t - 3.5) ** 2) / 12) return load, pv, price def generate_dataset( seed: int = 20260301, start: str = "2026-01-01", n_days: int = 120, peak_load_mw: float = 80.0, pv_capacity_mw: float = 30.0, adjustable_capacity_mw: float = 24.0, base_price: float = 400.0, ) -> dict: rng = np.random.default_rng(seed) load_p, pv_p, price_p = _profiles() d0 = date.fromisoformat(start) days = [] load_ar = 0.0 cloud = 0.8 for i in range(n_days): d = d0 + timedelta(days=i) weekend = d.weekday() >= 5 load_ar = 0.7 * load_ar + rng.normal(0, 0.04) day_factor = (0.85 if weekend else 1.0) * (1.0 + load_ar) load = peak_load_mw * load_p * day_factor * (1.0 + rng.normal(0, 0.02, INTERVALS)) load = np.clip(load, 0.0, None) cloud = float(np.clip(0.6 * cloud + 0.4 * rng.beta(4, 1.5), 0.05, 1.0)) pv = pv_capacity_mw * pv_p * cloud * (1.0 + rng.normal(0, 0.05, INTERVALS)) pv = np.clip(pv, 0.0, None) pv[pv_p == 0.0] = 0.0 net = (load - pv) / peak_load_mw price = base_price * (price_p + 0.35 * (net - net.mean())) * (1.0 + rng.normal(0, 0.03, INTERVALS)) if rng.random() < 0.06: # spike day: evening scarcity price[72:88] *= rng.uniform(1.6, 2.4) price = np.clip(price, 0.0, None) days.append( { "date": d.isoformat(), "weekend": weekend, "load_mw": [quantize(v, SCALE_MW) for v in load], "pv_mw": [quantize(v, SCALE_MW) for v in pv], "price_yuan_per_mwh": [quantize(v, SCALE_PRICE) for v in price], "adjustable_capacity_mw": [quantize(adjustable_capacity_mw, SCALE_MW)] * INTERVALS, } ) return { "meta": { "name": f"synthetic-hubei-v0-{start}-{n_days}d", "synthetic": True, "note": "SYNTHETIC shape-only placeholder. Replace with historical Hubei replay data; " "baselines on this set measure the harness, not the business KPI.", "generator": GENERATOR, "generator_version": GENERATOR_VERSION, "seed": seed, "interval_minutes": 15, "peak_load_mw": quantize(peak_load_mw, SCALE_MW), "pv_capacity_mw": quantize(pv_capacity_mw, SCALE_MW), "adjustable_capacity_mw": quantize(adjustable_capacity_mw, SCALE_MW), }, "days": days, } def main() -> None: ap = argparse.ArgumentParser(description="write the synthetic eval dataset") ap.add_argument("--out", required=True, type=Path) ap.add_argument("--seed", type=int, default=20260301) ap.add_argument("--days", type=int, default=120) args = ap.parse_args() ds = generate_dataset(seed=args.seed, n_days=args.days) args.out.parent.mkdir(parents=True, exist_ok=True) args.out.write_text(json.dumps(ds, indent=1) + "\n") print(f"wrote {args.out} ({len(ds['days'])} days)") if __name__ == "__main__": main()