# WashData - Home Assistant integration for appliance cycle monitoring via smart plugs. # Copyright (C) 2026 Lukas Bandura # SPDX-License-Identifier: AGPL-3.0-or-later # # This program is free software: you can redistribute it and/or modify # it under the terms of the GNU Affero General Public License as published # by the Free Software Foundation, either version 3 of the License, or # (at your option) any later version. # # This program is distributed in the hope that it will be useful, # but WITHOUT ANY WARRANTY; without even the implied warranty of # MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the # GNU Affero General Public License for more details. # # You should have received a copy of the GNU Affero General Public License # along with this program. If not, see . """Headless cycle-replay 'Playground' backend (Group F3). Pure, executor-safe logic behind the panel's Playground tab. Nothing here touches Home Assistant, fires events, or does I/O; the WebSocket handlers in ``ws_api.py`` call these helpers inside ``hass.async_add_executor_job``. Main entry points: - :func:`simulate_cycle_detail` - faithful single-cycle replay with per-step progress/remaining-time/phase/energy series and typed event log. - :func:`run_playground_history` - per-cycle rows + optional before/after diff. - :func:`run_playground_sweep` - objective 1D grid sweep. All top-level entry points are defensive: they never raise, returning an ``{"error": ...}`` marker instead so the WS handlers can relay it. """ from __future__ import annotations import logging import math from dataclasses import replace from datetime import datetime, timedelta, timezone from typing import Any, Callable import numpy as np from homeassistant.util import dt as dt_util from . import match_rules from . import notification_rules as notif_rules from . import progress as progress_mod from .options_utils import option_float, option_int from .signal_processing import ( compact_price_timeline, cycle_cost, energy_gap_threshold_s, integrate_wh, ) from .const import ( CONF_ANTI_WRINKLE_ENABLED, CONF_ANTI_WRINKLE_EXIT_POWER, CONF_ANTI_WRINKLE_IDLE_TIMEOUT, CONF_DISHWASHER_END_SPIKE_QUIET_RELEASE, CONF_SMART_TERMINATION_DURATION_RATIO, CONF_ANTI_CREASE_FINALIZE_RATIO, CONF_CURVE_PREROLL_SECONDS, CONF_ANTI_WRINKLE_MAX_DURATION, CONF_ANTI_WRINKLE_MAX_POWER, CONF_COMPLETION_MIN_SECONDS, CONF_END_ENERGY_THRESHOLD, CONF_PROFILE_MATCH_INTERVAL, CONF_PROFILE_MATCH_THRESHOLD, CONF_INTERRUPTED_MIN_SECONDS, CONF_LEARNING_CONFIDENCE, CONF_MATCH_PERSISTENCE, CONF_MIN_OFF_GAP, CONF_MIN_POWER, CONF_NOTIFY_ACTIONS, CONF_NOTIFY_BEFORE_END_MINUTES, CONF_NOTIFY_FINISH_SERVICES, CONF_NOTIFY_MILESTONES, CONF_NOTIFY_START_SERVICES, CONF_OFF_DELAY, CONF_PROFILE_MATCH_MAX_DURATION_RATIO, CONF_PROFILE_MATCH_MIN_DURATION_RATIO, CONF_PROFILE_UNMATCH_THRESHOLD, CONF_START_DURATION_THRESHOLD, CONF_START_THRESHOLD_W, CONF_STOP_THRESHOLD_W, CONF_WATCHDOG_INTERVAL, CYCLE_OVERRUN_ANOMALY_RATIO, CYCLE_UNDERRUN_ANOMALY_RATIO, DEFAULT_LEARNING_CONFIDENCE, DEFAULT_MATCH_PERSISTENCE, DEFAULT_MAX_DEFERRAL_SECONDS, DISHWASHER_END_SPIKE_WAIT_SECONDS, DEFAULT_NOTIFY_BEFORE_END_MINUTES, DEFAULT_NOTIFY_MILESTONES, DEFAULT_PROFILE_MATCH_MAX_DURATION_RATIO, DEFAULT_PROFILE_MATCH_MIN_DURATION_RATIO, DEFAULT_PROFILE_UNMATCH_THRESHOLD, STATE_ENDING, STATE_FINISHED, STATE_IDLE, STATE_OFF, STATE_RUNNING, STATE_STARTING, STATE_UNKNOWN, TerminationReason, resolve_watchdog_interval_default, ) from .cycle_detector import ( MatchContext, CycleDetector, CycleDetectorConfig, effective_anticrease_finalize_ratio, effective_curve_preroll_seconds, standby_near_stop_ceiling, terminal_high_for_guards, ) from .profile_store import ( MatchResult, ProfileStore, decompress_power_data, ) from .detector_config import ( terminal_drop_baseline_for, terminal_drop_enabled, terminal_drop_fires, terminal_drop_may_fire, ) from .time_utils import power_data_to_offsets _LOGGER = logging.getLogger(__name__) # The most recent N cycles to replay when the caller does not name any. DEFAULT_RECENT_CYCLES = 20 # Hard upper bound on cycles simulated in one batch call (defence in depth on # top of the caller-supplied ``concurrency`` cap). MAX_BATCH_CYCLES = 50 # Cap the per-cycle event log so a pathological trace cannot bloat the payload. MAX_EVENTS_PER_CYCLE = 300 # Cap the per-cycle timeline series so a very long cycle (4h dishwasher = ~2800 pts) # does not bloat the task result. Points are thinned at finalize time — evenly-spaced, # so the shape is preserved rather than truncated. MAX_SERIES_PER_CYCLE = 600 def _coerce_bool(value: Any) -> bool: """Strict bool coercion for override values. Plain ``bool()`` would read the string ``"false"`` as True, so a toggle sent as a string could switch a mode *on* when the user asked for it off. Unknown values raise, which ``build_sim_config`` turns into "ignore this override". """ if isinstance(value, bool): return value if isinstance(value, (int, float)): # Only the two values that actually mean a toggle. Anything else (2, -1, # NaN, inf) is a malformed override, not an intent to switch a mode on. if value == 0: return False if value == 1: return True raise ValueError(f"not a boolean: {value!r}") if isinstance(value, str): low = value.strip().lower() if low in ("true", "1", "yes", "on"): return True if low in ("false", "0", "no", "off"): return False raise ValueError(f"not a boolean: {value!r}") # Override keys the Playground honours, mapped to CycleDetectorConfig fields. # Only detection-relevant knobs matter; everything else in settings_override is # ignored safely. _OVERRIDE_FIELD_MAP: dict[str, tuple[str, Callable[[Any], Any]]] = { CONF_MIN_POWER: ("min_power", float), CONF_ANTI_WRINKLE_ENABLED: ("anti_wrinkle_enabled", _coerce_bool), CONF_ANTI_WRINKLE_MAX_POWER: ("anti_wrinkle_max_power", float), CONF_ANTI_WRINKLE_MAX_DURATION: ("anti_wrinkle_max_duration", float), CONF_ANTI_WRINKLE_EXIT_POWER: ("anti_wrinkle_exit_power", float), CONF_ANTI_WRINKLE_IDLE_TIMEOUT: ("anti_wrinkle_idle_timeout", float), CONF_DISHWASHER_END_SPIKE_QUIET_RELEASE: ("dishwasher_end_spike_quiet_release", float), CONF_SMART_TERMINATION_DURATION_RATIO: ("smart_termination_duration_ratio", float), CONF_ANTI_CREASE_FINALIZE_RATIO: ("anti_crease_finalize_ratio", float), CONF_CURVE_PREROLL_SECONDS: ("curve_preroll_seconds", float), CONF_OFF_DELAY: ("off_delay", int), CONF_MIN_OFF_GAP: ("min_off_gap", int), CONF_COMPLETION_MIN_SECONDS: ("completion_min_seconds", int), CONF_START_THRESHOLD_W: ("start_threshold_w", float), CONF_STOP_THRESHOLD_W: ("stop_threshold_w", float), CONF_START_DURATION_THRESHOLD: ("start_duration_threshold", float), CONF_INTERRUPTED_MIN_SECONDS: ("interrupted_min_seconds", int), # Suggested settings the Playground could not what-if (audit SUGGEST-19). CONF_END_ENERGY_THRESHOLD: ("end_energy_threshold", float), CONF_PROFILE_MATCH_THRESHOLD: ("match_confidence_threshold", float), CONF_PROFILE_MATCH_INTERVAL: ("match_interval", int), } # Matching options the Playground honours, mapped to the ``match_config`` key # they drive: the two Stage-1 duration ratios, both real user settings. The Stage # 2-4 scoring weights and DTW knobs were sandbox-only overrides until 0.5.8; they # could not persist, matching is saturated on them, and tuning them on 20 # in-sample cycles only overfit. Anything else in ``settings_override`` is ignored. _MATCH_OVERRIDE_KEYS: dict[str, tuple[str, Callable[[Any], Any]]] = { CONF_PROFILE_MATCH_MIN_DURATION_RATIO: ("min_duration_ratio", float), CONF_PROFILE_MATCH_MAX_DURATION_RATIO: ("max_duration_ratio", float), } # Canonical default for every matching override key, keyed by the OPTION key the # Playground uses. ``ws_get_constants`` ships it as ``pg_match_defaults`` and # ``effective_settings`` falls back to it for any key the live matcher config does # not carry. MATCH_DEFAULTS_BY_OPTION: dict[str, Any] = { CONF_PROFILE_MATCH_MIN_DURATION_RATIO: DEFAULT_PROFILE_MATCH_MIN_DURATION_RATIO, CONF_PROFILE_MATCH_MAX_DURATION_RATIO: DEFAULT_PROFILE_MATCH_MAX_DURATION_RATIO, } # Every option key the Playground control panel may carry (detection + matching). # Anything else submitted by a client is dropped. SETTING_KEYS: frozenset[str] = frozenset(_OVERRIDE_FIELD_MAP) | frozenset(_MATCH_OVERRIDE_KEYS) # The keys a user may publish from the Playground back into the live config. Every # Playground key is now a real config-entry option, so this is all of them; the # panel still gates its publish buttons on this list (shipped by # ``get_playground_settings``). PUBLISHABLE_SETTING_KEYS: frozenset[str] = SETTING_KEYS def effective_settings( base_config: CycleDetectorConfig, match_config: dict[str, Any] | None ) -> dict[str, Any]: """Option-keyed view of the values a simulation runs with when NO override is staged - i.e. the device's live, fully-resolved settings. The exact inverse of ``build_sim_config`` / ``apply_match_overrides``: it reads back the same fields those two write, so the Playground control panel shows the values the integration actually uses (device-type defaults included) instead of a static schema default that may have drifted. Never raises. """ out: dict[str, Any] = {} for opt_key, (field, coerce) in _OVERRIDE_FIELD_MAP.items(): value = getattr(base_config, field, None) if value is None: continue try: out[opt_key] = coerce(value) except (TypeError, ValueError, OverflowError): # pragma: no cover - defensive continue cfg = match_config or {} for opt_key, (cfg_key, coerce) in _MATCH_OVERRIDE_KEYS.items(): value = cfg.get(cfg_key, MATCH_DEFAULTS_BY_OPTION.get(opt_key)) if value is None: continue try: out[opt_key] = coerce(value) except (TypeError, ValueError, OverflowError): # pragma: no cover - defensive continue return out def sanitize_setting_values(values: Any) -> dict[str, Any]: """Filter a client-supplied settings map down to storable Playground values. Keeps only keys in :data:`SETTING_KEYS`, coerced with the same coercers the simulation uses, so an override can never carry an unknown key or a value that would be silently ignored at replay time. Never raises. """ if not isinstance(values, dict): return {} out: dict[str, Any] = {} for key, value in values.items(): if value is None: continue mapping = _OVERRIDE_FIELD_MAP.get(key) or _MATCH_OVERRIDE_KEYS.get(key) if mapping is None: continue _target, coerce = mapping try: coerced = coerce(value) # OverflowError: the override payload is JSON-decoded, so an oversized # integer literal arrives as an unbounded int and float() on one raises # rather than returning inf. Dropping the value is this function's # documented behaviour; escaping would fail the whole save. except (TypeError, ValueError, OverflowError): continue if isinstance(coerced, float) and not math.isfinite(coerced): continue # Every Playground setting is a physical quantity - watts, seconds, a # count, or a ratio - so a negative value is structurally meaningless and # would make the replayed detector behave in ways the live one never can # (e.g. an off_delay that expires before it starts). Rejected rather than # clamped: silently rewriting a value the user typed would make the sim # disagree with the control panel showing it back. if isinstance(coerced, (int, float)) and not isinstance(coerced, bool): if coerced < 0: continue out[key] = coerced return out def apply_match_overrides( match_config: dict[str, Any], settings_override: dict[str, Any] | None ) -> dict[str, Any]: """Return a copy of ``match_config`` with the recognised matching options from ``settings_override`` overlaid onto the matcher-config keys they drive. Unknown/None/malformed values are ignored, so a detection-only override leaves matching byte-identical to the live config.""" settings_override = sanitize_setting_values(settings_override) if not isinstance(settings_override, dict) or not settings_override: return match_config out = dict(match_config) for opt_key, (cfg_key, coerce) in _MATCH_OVERRIDE_KEYS.items(): val = settings_override.get(opt_key) if val is None: continue try: out[cfg_key] = coerce(val) except (TypeError, ValueError, OverflowError): pass return out def build_sim_config( base: CycleDetectorConfig, settings_override: dict[str, Any] | None ) -> CycleDetectorConfig: """Return a copy of ``base`` with the recognised override keys applied. Unknown keys and un-coercible values are ignored so a malformed override can never break a simulation. ``base`` is left untouched. """ settings_override = sanitize_setting_values(settings_override) if not isinstance(settings_override, dict) or not settings_override: return base changes: dict[str, Any] = {} for key, value in settings_override.items(): mapping = _OVERRIDE_FIELD_MAP.get(key) if mapping is None or value is None: continue field, coerce = mapping try: changes[field] = coerce(value) except (TypeError, ValueError, OverflowError): continue if not changes: return base try: return replace(base, **changes) except (TypeError, ValueError, OverflowError): # pragma: no cover - defensive return base def _cycle_base_time(cycle: dict[str, Any]) -> datetime: """Timezone-aware anchor for a cycle's offset-0 reading. Prefers the stored ISO ``start_time``; falls back to a fixed UTC epoch so offsets remain well-defined even for malformed cycles. """ raw = cycle.get("start_time") if isinstance(raw, datetime): return raw if raw.tzinfo else raw.replace(tzinfo=timezone.utc) if isinstance(raw, str) and raw: parsed = dt_util.parse_datetime(raw) if parsed is not None: return parsed if parsed.tzinfo else parsed.replace(tzinfo=timezone.utc) return datetime(2024, 1, 1, tzinfo=timezone.utc) def _cycle_label(cycle: dict[str, Any]) -> str | None: """The cycle's confirmed profile label (profile_name, else label).""" for key in ("profile_name", "label"): val = cycle.get(key) if isinstance(val, str) and val and val.lower() != "noise": return val return None # The grid a replay's snapshots are first built on: `resample_adaptive`'s floor, # which is what most cycles resolve to (it is max(5 s, the trace's median step)). _PLAYGROUND_START_DT = 5.0 # Bound on the keepalives emulated inside one silent stretch (8 h at a 30 s # watchdog): the detector's own 8 h cap ends any cycle long before this. _MAX_KEEPALIVES_PER_GAP = 960 def _build_match_snapshots( store: Any, ) -> tuple[list[dict[str, Any]], dict[str, Any], dict[str, list[str]], dict[str, Any]]: """Prepare the matcher snapshots + config once from the store. Mirrors the store's async matching path: one snapshot per profile using its sample cycle's decompressed trace, plus the store's live matching config (with any on-device tuned weight overrides merged in). Also resolves Stage-5 groups via :meth:`ProfileStore._grouped_snapshots`, the same call the live matcher makes. Note what that returns since #400: the **individual member** snapshots, unchanged, plus ``group_members`` and ``member_snaps``. It no longer averages a family into one ``__group__*`` aggregate - that averaged curve belonged to no member and cost the family its program-level match, so members are scored individually and each cohesive family is collapsed to its best member afterwards by :func:`collapse_group_candidates`. This docstring described the old aggregate behaviour long after the code stopped doing it. Returns ``(snapshots, match_config, group_members, member_snaps)``. When no cohesive groups exist ``group_members`` and ``member_snaps`` are both empty dicts and behaviour is identical to before. """ # in_progress: the sim replays a cycle step by step, so every match it runs is # a live one - the same footing as manager._async_do_perform_matching (#400). config = _matching_config(store, in_progress=True) # The live builder (item 387a), on the grid a replayed cycle starts on; # `_SimStore` re-grids per match, as live does. A store without it (the sim # run with no store at all) has nothing to match against. The hand-rolled # builder that used to serve a MagicMock store here had no production caller. if not callable(getattr(type(store), "build_match_snapshots", None)): return [], config, {}, {} try: snapshots = store.build_match_snapshots(_PLAYGROUND_START_DT) except Exception as exc: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground: live snapshot builder failed: %s", exc) snapshots = [] # Stage-5: map cohesive profile groups to their members; every member is # scored on its own curve and collapse_group_candidates forms the family. try: grouped_snaps, group_members, member_snaps = store._grouped_snapshots( # pylint: disable=protected-access snapshots ) except Exception as exc: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground: _grouped_snapshots failed: %s", exc) grouped_snaps, group_members, member_snaps = snapshots, {}, {} return grouped_snaps, config, group_members, member_snaps def _matching_config(store: Any, in_progress: bool = False) -> dict[str, Any]: """Live matcher config from the store (the live matcher runs no overrides).""" return { "min_duration_ratio": float(getattr(store, "_min_duration_ratio", DEFAULT_PROFILE_MATCH_MIN_DURATION_RATIO)), "max_duration_ratio": float(getattr(store, "_max_duration_ratio", DEFAULT_PROFILE_MATCH_MAX_DURATION_RATIO)), "dtw_bandwidth": float(getattr(store, "dtw_bandwidth", 0.2)), # Mirror the live Stage-4 energy discriminator so the sim is byte-identical. "energy_mode": str(getattr(store, "energy_mode", "mean")), "in_progress": bool(in_progress), } class _InlineExecutor: """``hass`` for :class:`_SimStore`: an executor job runs inline, in this thread. The replay is already in an executor thread (or a harness with no loop), and the store's real ``hass`` belongs to the event loop, so it must not be used. """ async def async_add_executor_job(self, fn: Callable[..., Any], *args: Any) -> Any: return fn(*args) def _run_inline(coro: Any) -> Any: """Drive a store coroutine whose only awaits are :class:`_InlineExecutor` jobs. Those complete without suspending, so the coroutine finishes on its first step. If a future change gives it a real suspension point this raises instead of returning a wrong answer, and the replay reports the failure. """ try: coro.send(None) except StopIteration as stop: return stop.value coro.close() raise RuntimeError("store coroutine suspended; the Playground cannot await it") #: Distinct query grids a replay keeps candidate templates for. _MAX_SNAPSHOT_GRIDS = 64 #: How long the synthetic tail may wait out a held verified pause: the watchdog's #: own limit for silence under one (``manager._watchdog_check_stuck_cycle``). _TAIL_VERIFIED_PAUSE_CAP_S = DEFAULT_MAX_DEFERRAL_SECONDS + 1800.0 def _tail_span_s(config: Any) -> float: """How long the synthetic 0 W tail runs so a natural end can fire. Past the longest ordinary end gate (off delay / min off gap), plus margin. A dishwasher also waits up to ``DISHWASHER_END_SPIKE_WAIT_SECONDS`` for a late pump-out, so its tail covers that: sized on the two settings alone, a what-if that lowered them (as Apply all does) force-stopped a cycle the detector would have ended normally (one Eco cycle needed 1530 s against a 600 s tail; found by devtools/suggestion_loop_eval.py, register item 455). """ gate = max(float(config.off_delay or 0.0), float(config.min_off_gap or 0.0)) if getattr(config, "device_type", None) == "dishwasher": gate = max(gate, DISHWASHER_END_SPIKE_WAIT_SECONDS) return gate * 1.5 + 300.0 class _SimStore: """The device's store as the live matcher sees it, callable from a replay. The replay runs the REAL ``ProfileStore.async_match_profile`` and ``ProfileStore.async_verify_alignment`` with this object as ``self`` (item 387a, audit PLAYGROUND-01): every attribute not defined here is the store's own, so the candidate pool, Stage 1-5 (incl. the in-progress member preference), the 12-point floor, ambiguity, the prefix flags, the member confidence, the phase lookup and the alignment thresholds are the live code rather than a copy that can drift. Three things differ, all deliberate: * ``hass`` runs executor jobs inline (:class:`_InlineExecutor`); * the matcher config is the sim's - the live config plus any what-if override of the Stage-1 ratios - returned from ``_matching_overrides``, which ``async_match_profile`` merges last; * candidate templates are cached per query grid, seeded with the prebuilt 5 s set a batch shares. A store without the live builder (a MagicMock in tests) gets the prebuilt set whatever the grid, as before. """ def __init__( self, store: Any, match_config: dict[str, Any], prebuilt: tuple[Any, Any, Any, Any], ) -> None: self._store = store self.hass = _InlineExecutor() self._config = {k: v for k, v in (match_config or {}).items() if k != "in_progress"} self.dtw_bandwidth = self._config.get( "dtw_bandwidth", getattr(store, "dtw_bandwidth", 0.2) ) snaps, _cfg, group_members, member_snaps = prebuilt self._prebuilt = (snaps, (snaps, group_members or {}, member_snaps or {})) self._live_builder = callable(getattr(type(store), "build_match_snapshots", None)) self._grids: dict[float, tuple[Any, Any]] = ( {float(_PLAYGROUND_START_DT): self._prebuilt} if self._live_builder else {} ) self._pending: tuple[Any, Any] | None = None def __getattr__(self, name: str) -> Any: return getattr(self._store, name) def _matching_overrides(self) -> dict[str, Any]: return dict(self._config) def build_match_snapshots(self, used_dt: float) -> list[dict[str, Any]]: if not self._live_builder: self._pending = self._prebuilt return self._prebuilt[0] key = float(used_dt) hit = self._grids.get(key) if hit is None: snaps = self._store.build_match_snapshots(used_dt) hit = (snaps, self._store._grouped_snapshots(snaps)) # noqa: SLF001 if len(self._grids) >= _MAX_SNAPSHOT_GRIDS: self._grids.clear() self._grids[key] = hit self._pending = hit return hit[0] def _grouped_snapshots(self, snapshots: list[dict[str, Any]]) -> Any: pending = self._pending if pending is not None and pending[0] is snapshots: return pending[1] return self._store._grouped_snapshots(snapshots) # noqa: SLF001 def match( self, readings: Any, duration: float, in_progress: bool = False, stop_threshold_w: float | None = None, ) -> MatchResult: """``ProfileStore.async_match_profile`` on this view, run to completion.""" return _run_inline( ProfileStore.async_match_profile( self, readings, duration, # type: ignore[arg-type] in_progress=in_progress, stop_threshold_w=stop_threshold_w, ) ) def verify_alignment(self, profile_name: str, trace: Any) -> tuple[bool, float, float]: """``ProfileStore.async_verify_alignment`` on this view, run to completion.""" return _run_inline( ProfileStore.async_verify_alignment(self, profile_name, trace) # type: ignore[arg-type] ) def _readings_from_cycle( cycle: dict[str, Any], ) -> tuple[list[tuple[datetime, float]], list[tuple[float, float]], datetime]: """Reconstruct (datetime, power) readings + (offset, power) points + base time.""" points = decompress_power_data(cycle) base = _cycle_base_time(cycle) readings = [(base + timedelta(seconds=float(o)), float(p)) for o, p in points] return readings, points, base # ─── Single-cycle faithful simulation (Simulate mode) ─────────────────────────── # States in which no progress estimate is shown (mirrors _update_remaining_only). _DEAD_STATES = (STATE_OFF, STATE_UNKNOWN, STATE_IDLE) _SIM_SERIES_THROTTLE_S = 30.0 # cap estimator calls; 5s matched cadence made this a no-op # Terminal-drop baselines per stored-cycle list (audit ML-08). A History/Optimize # batch replays many cycles against one store and the baseline decompresses every # completed trace, so it is built once per list. The list is held, so its id # cannot be recycled while cached; an append changes the length in the key. _TERMINAL_DROP_BASELINES: dict[tuple[int, int, float], tuple[Any, Any]] = {} _MAX_TERMINAL_DROP_BASELINES = 16 def _sim_terminal_drop_baseline( store: Any, stop_threshold_w: float ) -> tuple[float | None, tuple[float, float] | None]: """The live terminal-drop baseline over ``store``'s stored cycles. Never raises. Built from every stored cycle, the replayed one included, like the rest of the Playground (in-sample). That can only make a completed cycle LESS likely to fire: its own first quiet span is in the baseline. """ try: cycles = store.get_past_cycles() if not isinstance(cycles, list): return None, None key = (id(cycles), len(cycles), float(stop_threshold_w)) hit = _TERMINAL_DROP_BASELINES.get(key) if hit is not None and hit[0] is cycles: return hit[1] baseline = terminal_drop_baseline_for(list(cycles), stop_threshold_w) if len(_TERMINAL_DROP_BASELINES) >= _MAX_TERMINAL_DROP_BASELINES: _TERMINAL_DROP_BASELINES.clear() _TERMINAL_DROP_BASELINES[key] = (cycles, baseline) return baseline except Exception: # pylint: disable=broad-exception-caught return None, None def simulate_cycle_detail( cycle: dict[str, Any], base_config: CycleDetectorConfig, settings_override: dict[str, Any] | None, store: Any, options: dict[str, Any] | None, price: float | None = None, compute_series: bool = True, prebuilt: tuple[Any, Any, Any, Any] | None = None, ) -> dict[str, Any]: """Faithful single-cycle replay for the Playground "Simulate" view. Drives the REAL :class:`CycleDetector`, the real matcher and the manager's :mod:`match_rules` over the cycle's own trace, and calls the SAME :mod:`progress` and :mod:`notification_rules` functions the live integration uses (the estimator every ``_SIM_SERIES_THROTTLE_S`` = 30 s of replay time, with the live per-second EMA scaling) - so the timeline is what would happen live. No detection/progress/notification math is implemented here; this only orchestrates the shared code. Never raises; returns ``{"error": ...}`` on failure. Read-only: nothing is persisted and no notifications are sent. Returns ``{cycle_id, label, duration_s, config_summary, series, events, alerts, outcome}`` (see the design doc for the field contract). """ options = options or {} try: return _simulate_cycle_detail_inner( cycle, base_config, settings_override, store, options, price, compute_series, prebuilt, ) except Exception as exc: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground detail sim failed for %s: %s", cycle.get("id"), exc) return {"error": str(exc), "cycle_id": cycle.get("id")} def _device_type_of(config: CycleDetectorConfig) -> str: return getattr(config, "device_type", "washing_machine") def build_cycle_detail_sim_by_id( store: Any, cycle_id: str, base_config: CycleDetectorConfig, settings_override: dict[str, Any] | None, options: dict[str, Any] | None, price: float | None = None, ) -> "_DetailSim | dict[str, Any]": """Look up a stored cycle by id and build a resumable :class:`_DetailSim`. Used by the chunked background-task driver in ``ws_api`` so the heavy replay can be stepped across many small executor jobs (issue #311). Returns a ``{"error": ...}`` marker (not a sim) when the id is unknown or setup fails, so the caller can surface it. The store lookup + build run together so the WS handler can offload the whole thing to an executor thread. Never raises.""" options = options or {} try: cycle = next( (c for c in store.get_past_cycles() if c.get("id") == cycle_id), None ) except Exception as exc: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground detail lookup failed for %s: %s", cycle_id, exc) return {"error": str(exc), "cycle_id": cycle_id} if cycle is None: return {"error": "not_found", "cycle_id": cycle_id} try: return _DetailSim(cycle, base_config, settings_override, store, options, price) except Exception as exc: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground detail sim build failed for %s: %s", cycle_id, exc) return {"error": str(exc), "cycle_id": cycle_id} def _simulate_cycle_detail_inner( cycle: dict[str, Any], base_config: CycleDetectorConfig, settings_override: dict[str, Any] | None, store: Any, options: dict[str, Any], price: float | None, compute_series: bool = True, prebuilt: tuple[Any, Any, Any, Any] | None = None, ) -> dict[str, Any]: """One-shot faithful replay: build the resumable sim and run it to completion. The chunked (background-task) driver in ``ws_api`` builds the same :class:`_DetailSim` and calls ``step``/``run_tail``/``finalize`` across many small executor jobs so the event loop breathes on very long cycles (issue #311). Because both paths drive the identical object in the identical order, the timeline is byte-for-byte the same (tests/test_playground_chunked_parity.py). """ sim = _DetailSim( cycle, base_config, settings_override, store, options, price, compute_series, prebuilt, ) if not sim.ready: return sim.empty_payload() sim.step(0, sim.n_readings) sim.run_tail() return sim.finalize() class _DetailSim: """Resumable single-cycle Playground "Simulate" replay. Drives the REAL :class:`CycleDetector` + the real matcher (``ProfileStore.async_match_profile``, via :class:`_SimStore`) over the cycle's own trace, applies each match with the manager's own rules (:mod:`match_rules`: switching, verified pause, confident-mismatch revoke) and calls the SAME :mod:`progress` and :mod:`notification_rules` functions the live integration uses. No detection/matching/progress/notification math is implemented here; this only orchestrates the shared code. Read-only: nothing is persisted and no notifications are sent. The replay is split into :meth:`step` (a slice of the real readings), :meth:`run_tail` (the synthetic quiet tail + flush) and :meth:`finalize` (outcome + alerts) so a long cycle can be replayed chunk-by-chunk across executor jobs without holding the GIL for the whole run. """ def __init__( self, cycle: dict[str, Any], base_config: CycleDetectorConfig, settings_override: dict[str, Any] | None, store: Any, options: dict[str, Any], price: float | None, compute_series: bool = True, prebuilt: tuple[Any, Any, Any, Any] | None = None, ) -> None: self.cycle = cycle self.store = store self.options = options or {} self.price = price # Dynamic tariff timeline frozen onto the stored cycle (#426). Replaying it # is what keeps the sim's projected cost identical to the one the live # estimator produced for that cycle; without it a dynamically-priced cycle # would replay at a single flat price and silently diverge. self.price_points = compact_price_timeline( [ (entry[0], entry[1]) for entry in (cycle.get("price_timeline") or []) if isinstance(entry, (list, tuple)) and len(entry) >= 2 ] ) self.compute_series = compute_series self.config = build_sim_config(base_config, settings_override) self.device_type = _device_type_of(self.config) self.label = _cycle_label(cycle) self.readings, _points, self.base = _readings_from_cycle(cycle) self.stored_duration = _safe_float(cycle.get("duration")) # When the appliance actually ran, for the Optimize end-timing objectives # (audit PLAYGROUND-08): first/last reading above the LIVE stop threshold, # never the override's, so a swept threshold cannot move its own yardstick. # Same definition as devtools/end_gate_eval.py (`_active_span`). _truth_stop = float(getattr(base_config, "stop_threshold_w", 0.0) or 0.0) _active = [ (ts - self.base).total_seconds() for ts, p in self.readings if p > _truth_stop ] self.active_end_s: float | None = _active[-1] if _active else None self.active_span_s: float | None = ( _active[-1] - _active[0] if len(_active) >= 2 else None ) self.outcome: dict[str, Any] = { "detected": False, "detected_count": 0, "termination_reason": None, "status": None, "final_duration_s": None, "matched_profile": None, "match_correct": None, "confidence": None, "expected_s": None, "overrun_ratio": None, "projected_energy_wh": None, "projected_cost": None, # Would the manager auto-label the (primary) finished cycle, as which # profile, and why not: match_rules.cycle_end_label_verdict reasons # (ok / no_winner / below_floor / ambiguous / margin / unknown_profile), # plus no_cycle / no_store / error from the replay itself. "would_label": False, "label_profile": None, "label_reason": "no_cycle", "end_offset_s": None, } if prebuilt is None: prebuilt = _build_match_snapshots(store) # Overlay any matcher-knob overrides. Because history/sweep run through this # same class, a swept matching value flows in via settings_override too; # applying to a copy keeps the shared prebuilt match_config untouched. self.match_config = apply_match_overrides(prebuilt[1], settings_override) # The live matcher and alignment check, run inline (see _SimStore). self.view: _SimStore | None = ( _SimStore(store, self.match_config, prebuilt) if store is not None else None ) self.ready = len(self.readings) >= 5 # Per-sim end-expectation cache, threaded through the shared progress helpers # exactly like the manager threads self._ml_end_expectation_cache. self.endexp_cache: list[Any] = [None] self.events: list[dict[str, Any]] = [] self.series: list[dict[str, Any]] = [] self.captured: list[dict[str, Any]] = [] self.cursor = {"t": 0.0} self.last_match: dict[str, Any] = { "name": None, "conf": 0.0, "ambiguous": False, "expected": 0.0, } self.last_logged = {"kind": None, "name": None} # The manager's switching state and the options its rules read, resolved # exactly as `manager._load_runtime_options` resolves them (audit # PLAYGROUND-03): the reported program is the one live would display. self.switch = match_rules.SwitchState() self.match_persistence = option_int( self.options.get(CONF_MATCH_PERSISTENCE, DEFAULT_MATCH_PERSISTENCE), DEFAULT_MATCH_PERSISTENCE, minimum=1, ) self.unmatch_threshold = self.options.get( CONF_PROFILE_UNMATCH_THRESHOLD, DEFAULT_PROFILE_UNMATCH_THRESHOLD ) self.learning_floor = float( option_float( self.options.get(CONF_LEARNING_CONFIDENCE, DEFAULT_LEARNING_CONFIDENCE), DEFAULT_LEARNING_CONFIDENCE, ) or 0.0 ) # The last live MatchResult (manager._last_match_result) and, per finished # cycle, what the manager held when it ended (its cycle-end inputs). self._last_result: Any = None self.cycle_ends: list[dict[str, Any]] = [] self.smoothed: dict[str, Any] = {"v": 0.0, "program": None} self.flags = {"detected": False, "pre_complete": False, "start": False} # --- notification config (decisions reuse notification_rules) --- self.start_configured = bool( self.options.get(CONF_NOTIFY_START_SERVICES) or self.options.get(CONF_NOTIFY_ACTIONS) ) self.finish_configured = bool( self.options.get(CONF_NOTIFY_FINISH_SERVICES) or self.options.get(CONF_NOTIFY_ACTIONS) ) self.before_end = float( self.options.get(CONF_NOTIFY_BEFORE_END_MINUTES, DEFAULT_NOTIFY_BEFORE_END_MINUTES) or 0.0 ) self.quiet_bounds = notif_rules.quiet_hours_bounds(self.options) self.last_sample_t = -1e9 self._aborted = False # Watchdog cadence (item 390). Live injects a keepalive on this cadence # while a cycle sits below the stop threshold and the plug is silent; the # sim does the same inside a silent stretch of the trace (see step()). try: _wd = float( {**self.options, **(settings_override or {})}.get( CONF_WATCHDOG_INTERVAL, resolve_watchdog_interval_default(self.device_type), ) ) except (TypeError, ValueError, OverflowError): _wd = float(resolve_watchdog_interval_default(self.device_type)) self.watchdog_s = _wd if math.isfinite(_wd) and _wd > 0 else 0.0 self._last_real: tuple[datetime, float] | None = None if self.ready: self.detector = CycleDetector( self.config, self._on_state_change, self._on_cycle_end, profile_matcher=self._matcher, device_name="playground-detail", terminal_drop_provider=self._terminal_drop_provider( {**self.options, **(settings_override or {})} ), ) def _terminal_drop_provider( self, options: dict[str, Any] ) -> Callable[[list[tuple[float, float]], float], bool] | None: """``manager._terminal_drop_provider`` for this replay; None where live would not run it (``detector_config.terminal_drop_enabled``, audit ML-08).""" if not terminal_drop_enabled(self.device_type, options): return None stop = float(getattr(self.config, "stop_threshold_w", 0.0) or 0.0) baseline = _sim_terminal_drop_baseline(self.store, stop) device_type = self.device_type def _provider(points: list[tuple[float, float]], _expected: float) -> bool: # The live gate on the sim's own detector (built after this closure). if not terminal_drop_may_fire(device_type, options, self.detector): return False return terminal_drop_fires(points, baseline, stop) return _provider @property def n_readings(self) -> int: return len(self.readings) def empty_payload(self) -> dict[str, Any]: return { "cycle_id": self.cycle.get("id"), "label": self.label, "duration_s": self.stored_duration, "start_time": self.cycle.get("start_time"), "active_end_s": _safe_float(self.active_end_s), "active_span_s": _safe_float(self.active_span_s), "config_summary": _sim_config_summary(self.config), "series": [], "events": [], "alerts": [], "outcome": self.outcome, } def _end_exp_fn(self, name: str, dur: float) -> Any: exp, self.endexp_cache[0] = progress_mod.profile_end_expectation( self.store, name, dur, self.endexp_cache[0] ) return exp def _emit(self, etype: str, detail: str, severity: str = "info") -> None: if len(self.events) < MAX_EVENTS_PER_CYCLE: self.events.append( {"t": round(self.cursor["t"], 1), "type": etype, "detail": detail, "severity": severity} ) def _held(self, offset: float) -> bool: # Quiet hours are local clock hours and replay timestamps are UTC, so # `.hour` on the raw stamp held the wrong hours (audit PROGRESS-13). return notif_rules.in_quiet_hours( self.quiet_bounds, dt_util.as_local(self.base + timedelta(seconds=offset)) ) def _on_state_change(self, old_state: str, new_state: str) -> None: self._emit("state", f"{old_state}->{new_state}") # A new cycle: the switching state starts fresh, exactly when # `manager._on_state_change` resets it (RUNNING from OFF / STARTING / # UNKNOWN). PAUSED/ENDING -> RUNNING is a resume and keeps it. if new_state == STATE_RUNNING and old_state in (STATE_OFF, STATE_STARTING, STATE_UNKNOWN): self.switch.start_cycle() self.last_match.update(name=None, conf=0.0, expected=0.0, ambiguous=False) self.last_logged.update(kind=None, name=None) if ( not self.flags["detected"] and new_state == STATE_RUNNING and old_state in (STATE_OFF, STATE_UNKNOWN, STATE_STARTING, STATE_IDLE) ): self.flags["detected"] = True self._emit("detected", "cycle detected (running)") if self.start_configured and not self.flags["start"]: self.flags["start"] = True # Start notifications are never delayed by quiet hours (live # contract), so the sim always emits them immediately. self._emit("notify_start", "start notification") def _on_cycle_end(self, cycle_data: dict[str, Any]) -> None: self.captured.append(cycle_data) reason = cycle_data.get("termination_reason") self._emit("finished", f"reason={reason} status={cycle_data.get('status')}", "info") # What `manager._async_process_cycle_end` freezes before its first await: # the displayed program and the last live match (its label inputs). st = self.switch program = st.current_program if match_rules.program_is_committed(st.current_program) else None self.cycle_ends.append({ "program": program, "confidence": st.last_confidence if program else None, "expected": st.matched_duration if program else None, "ambiguous": bool(self.last_match.get("ambiguous")), "live_result": self._last_result, # Replay offset at which WashData said "done" (end lag, PLAYGROUND-08). "t": self.cursor["t"], }) # ...and the terminal reset at its tail: the next cycle starts from "off" # with no live result, as live does once the cycle has been processed. st.current_program = "off" st.matched_duration = None self._last_result = None def _has_real_profiles(self) -> bool: """The gate `manager._async_perform_combined_matching` checks first.""" if self.store is None: return False try: return bool(self.store.has_real_profiles) except Exception: # pylint: disable=broad-exception-caught return False def _verify_alignment( self, profile_name: str, det_readings: list[tuple[datetime, float]] ) -> tuple[bool, float]: """The live alignment check; a failure counts as unconfirmed, as live.""" try: assert self.view is not None formatted = power_data_to_offsets(det_readings) # type: ignore[arg-type] is_confirmed, mapped_time, _ = self.view.verify_alignment(profile_name, formatted) return bool(is_confirmed), mapped_time except Exception: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground alignment check failed", exc_info=True) return False, 0.0 def _get_profile(self, name: str) -> Any: return self.store.get_profile(name) def _matcher(self, det_readings: list[tuple[datetime, float]]) -> Any: """One live match tick: ``manager._async_do_perform_matching`` in sequence. The real matcher (``ProfileStore.async_match_profile`` via :class:`_SimStore`), then the manager's own rules from :mod:`match_rules`: switching, the envelope verified pause and its releases, the consistency override (which is also how a confident mismatch drops the program). Like the manager it pushes the verified pause and the commit flag to the detector, and hands it the tick's name - which after a divergence revert is "detecting...", as live - not the displayed program. Returns the match context the detector applies, or None where live would not have matched at all (no real profiles). The alignment check runs synchronously here. Live awaits it, and the matcher, while readings keep arriving; the replay applies the tick at the reading that triggered it. """ if not det_readings or not self._has_real_profiles(): return None assert self.view is not None det = self.detector current_duration = (det_readings[-1][0] - det_readings[0][0]).total_seconds() stop_w = float(det.config.stop_threshold_w) result = self.view.match( det_readings, current_duration, in_progress=True, stop_threshold_w=stop_w ) self._last_result = result st = self.switch prev_program = st.current_program tick = match_rules.begin_tick(st, result, self.match_persistence, current_duration) match_rules.decide_switch( st, tick, result, self.match_persistence, self.unmatch_threshold ) match_rules.record_scores(st, result.candidates) current_matched = det.matched_profile prev_verified = getattr(det, "_verified_pause", False) current_power = det_readings[-1][1] # The detector mirrors the manager's user pause (`set_user_paused`); a replay # never pauses, but a harness can (`end_gate_eval --user-pause`, item 514). user_paused = getattr(det, "_user_paused", False) is True alignment: tuple[bool, float] | None = None if match_rules.needs_alignment_check(current_matched, current_power, stop_w, user_paused): alignment = self._verify_alignment(current_matched, det_readings) pause = match_rules.decide_alignment_pause( verified_pause=prev_verified, current_matched=current_matched, alignment=alignment, envelope_span=self.view.envelope_time_span, ) pause = match_rules.decide_pause_release( verified_pause=pause.verified_pause, current_matched=current_matched, current_power=current_power, stop_threshold_w=getattr(det.config, "stop_threshold_w", 5.0), user_paused=user_paused, expected_duration=det.expected_duration_seconds, current_duration=current_duration, time_below=getattr( det, "_time_below_threshold_gapfree", getattr(det, "_time_below_threshold", 0.0) ), program=st.current_program, ) verified = pause.verified_pause match_rules.consistency_override(st, tick, result, verified, self._get_profile) # The manager's ENDING pause hold (register item 469b). verified = match_rules.hold_in_ending( ending=det.state == STATE_ENDING, is_ambiguous=bool(result.is_ambiguous), current_matched=current_matched, prev_verified=prev_verified, verified_pause=verified, user_paused=user_paused, ).verified_pause det.set_verified_pause(verified) det.set_match_committed(match_rules.program_is_committed(st.current_program)) self._report_tick(prev_program, bool(prev_verified), bool(verified), result) return self._match_context(tick, tick.phase_name, result) def _report_tick( self, prev_program: Any, prev_verified: bool, verified: bool, result: Any ) -> None: """Events + the reported match after a tick: what live would display.""" st = self.switch program = st.current_program if match_rules.program_is_committed(st.current_program) else None before = prev_program if match_rules.program_is_committed(prev_program) else None if program != before: if program and not before: self._emit("match_commit", f"{program} (conf={float(st.last_confidence):.2f})") elif program: self._emit( "match_changed", f"{before} -> {program} (conf={float(st.last_confidence):.2f})", ) else: self._emit("match_reverted", f"{before} -> detecting") self.last_logged.update(kind="matched" if program else "reverted", name=program) if result.is_confident_mismatch: if self.last_logged["kind"] != "unmatched": self._emit("unmatched", "no candidate") self.last_logged["kind"] = "unmatched" elif result.is_ambiguous and not program and result.best_profile: # Ambiguous before any commit: stay 'detecting', surface it once per name. raw = result.best_profile if self.last_logged["kind"] != "ambiguous" or self.last_logged["name"] != raw: cands = result.candidates or [] runner = cands[1].get("name") if len(cands) > 1 else None self._emit( "match_ambiguous", f"{raw} vs {runner} (margin={float(result.ambiguity_margin):.3f})", "warn", ) self.last_logged.update(kind="ambiguous", name=raw) if verified != prev_verified: self._emit("verified_pause", "engaged" if verified else "released") self.last_match.update( name=program, conf=float(st.last_confidence or 0.0) if program else 0.0, expected=float(st.matched_duration or 0.0) if program else 0.0, # The raw tick's flag, as `manager._last_match_ambiguous` holds it. ambiguous=bool(result.is_ambiguous), ) def _match_context(self, tick: Any, phase_name: str | None, result: Any) -> MatchContext: """The named context the manager builds for ``update_match`` (DETECT-15).""" store = self.store det = self.detector # The manager's own rule: a divergence revert revokes the detector's match. name, revoke = match_rules.detector_match(tick, result) def _ask(fn: Callable[[], Any]) -> Any: # Guarded like the rest of the sim: a partial test double without one # of these methods must not turn every match into "unmatched". try: return fn() except Exception: # pylint: disable=broad-exception-caught return None stop_w = float(det.config.stop_threshold_w) return MatchContext( profile_name=name, confidence=tick.confidence, expected_duration=tick.matched_duration, phase_name=phase_name, is_confident_mismatch=revoke, is_ambiguous=result.is_ambiguous, is_prefix_ambiguous_full_shape=result.is_prefix_ambiguous_full_shape, tail_power=_ask(lambda: store.profile_tail_power(name)) if name else None, # One implementation with the manager's `_terminal_high_for_guards`. terminal_high=_ask(lambda: terminal_high_for_guards( store, det.config, getattr(det, "_cycle_max_power", 0.0), name )), terminal_quiet_s=( _ask(lambda: store.profile_terminal_quiet_seconds(name)) if name else None ), longest_candidate_s=float(getattr(result, "longest_candidate_duration_s", 0.0) or 0.0), trusted_min_s=( _ask(lambda: store.profile_trusted_min_duration(name)) if name else None ), pause_catalogue=( _ask(lambda: store.profile_pause_catalogue(name, stop_w)) if name else None ), # #452, as the manager supplies it (the stall display and its # standby-band hold). stall_catalogue=( (lambda: _ask(lambda: store.profile_pause_catalogue( name, standby_near_stop_ceiling(stop_w) ))) if name else None ), ) def _label_decision(self, cycle_data: dict[str, Any], live_result: Any) -> tuple[str | None, str]: """Would the manager auto-label this finished cycle? ``(profile, reason)``. The manager's cycle-end path: ONE complete-cycle match on the stored trace (``match_rules.final_match_input`` + ``async_match_profile``, not in progress), falling back to the last live match when the trace is too short, then ``match_rules.cycle_end_label_verdict`` at the learning floor. """ if self.view is None: return None, "no_store" try: final = None final_input = match_rules.final_match_input(cycle_data) if final_input is not None: final = self.view.match(final_input[0], final_input[1]) match_result = final if final is not None else live_result return match_rules.cycle_end_label_verdict( match_result, self.learning_floor, self.store.get_profiles() ) except Exception: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground label decision failed", exc_info=True) return None, "error" def _price_at(self, offset_s: float) -> float | None: """The tariff in force at a replay offset, or the flat price with no timeline. Live charges the *remaining* energy at whatever ``_resolve_energy_price()`` returns at that instant, so a replay has to move through its own stored timeline instead of pinning the whole cycle to one price. Otherwise a dynamically-priced cycle diverges from the projection the live estimator actually produced, which is the one thing this sim exists to reproduce. """ price = self.price for point_offset, point_price in self.price_points: if point_offset > offset_s: break price = point_price return price def _cost_so_far( self, trace: list[tuple[datetime, float]] ) -> tuple[float, float] | None: """``(cost, charged_wh)`` incurred up to this point of the replay, or None. Mirrors ``manager._live_cost_so_far``: the integrated trace charged at the prices the cycle actually ran through. None (no stored timeline) puts the projection back on the flat-price formula, which is the right answer for a cycle recorded before dynamic pricing existed. """ if not self.price_points or len(trace) < 2: return None try: base_ts = self.base.timestamp() timestamps = np.asarray([t.timestamp() - base_ts for t, _ in trace], dtype=float) power = np.asarray([p for _, p in trace], dtype=float) max_gap_s = energy_gap_threshold_s(timestamps) result = cycle_cost(timestamps, power, self.price_points, max_gap_s=max_gap_s) if result is None: return None return result[0], float(integrate_wh(timestamps, power, max_gap_s=max_gap_s)) except Exception: # noqa: BLE001 - the sim never raises return None def _sample(self, ts: datetime) -> None: if not self.compute_series: return # batch/sweep rows only need the outcome, not the per-step series offset = (ts - self.base).total_seconds() if offset - self.last_sample_t < _SIM_SERIES_THROTTLE_S: return prev_sample_t = self.last_sample_t self.last_sample_t = offset state = self.detector.state power = 0.0 trace = self.detector.get_power_trace() if trace: power = float(trace[-1][1]) energy_wh = float(getattr(self.detector, "_energy_since_idle_wh", 0.0) or 0.0) pt: dict[str, Any] = { "t": round(offset, 1), "power": round(power, 1), "energy_wh": round(energy_wh, 2), "state": state, "progress": None, "remaining_s": None, "phase": None, "confidence": round(self.last_match["conf"], 3) if self.last_match["name"] else None, "matched_profile": self.last_match["name"], } if getattr(self.detector, "stalled", False) is True: pt["stalled"] = True # #452: shown as paused / Stalled live matched_dur = float(self.last_match["expected"] or 0.0) program = self.last_match["name"] if state not in _DEAD_STATES and program and matched_dur > 0: # Item 514, as live: progress reads programme time (a halt is not # progress); the energy and the cost below read the whole trace. prog_trace = self.detector.progress_trace(ts) prog_t = self.detector.progress_elapsed_s(offset, ts) phase_result = None if len(prog_trace) >= 10 and program != "detecting...": phase_result = progress_mod.estimate_phase_progress( self.store, prog_trace, prog_t, program, quiet_threshold_w=float( getattr(self.detector.config, "stop_threshold_w", 0.0) or 0.0 ), ) result = progress_mod.compute_progress( self.device_type, matched_dur, prog_t, progress_mod.ema_seed(self.smoothed["v"], self.smoothed["program"], program), phase_result, # Same time-scaled smoothing as live: the sim steps the estimator # at its own throttle, so without this the replay would smooth # over 30 s steps as if they were the manager's 5 s ones. dt_seconds=( offset - prev_sample_t if prev_sample_t >= 0.0 else None ), ) if result is not None: self.smoothed["v"] = result.smoothed self.smoothed["program"] = program pt["progress"] = round(result.progress, 1) pt["remaining_s"] = round(result.remaining, 0) pt["phase"] = progress_mod.current_phase( self.store, state, program, result.progress, matched_dur ) sim_cost = self._cost_so_far(trace) wh, cost = progress_mod.projected_energy( self.store, self.options, matched_dur, trace, program, result.progress, energy_wh, self._price_at(offset), self._end_exp_fn, cost_so_far=sim_cost[0] if sim_cost else None, cost_so_far_wh=sim_cost[1] if sim_cost else None, ) pt["projected_energy_wh"] = round(wh, 1) if wh is not None else None pt["projected_cost"] = round(cost, 4) if cost is not None else None # One-time pre-completion marker (reuses the production predicate). if not self.flags["pre_complete"] and notif_rules.should_notify_pre_completion( self.before_end, self.flags["pre_complete"], result.remaining, result.progress, self.last_match["ambiguous"], ): self.flags["pre_complete"] = True held = self._held(offset) self._emit( "notify_held" if held else "notify_pre_complete", "pre-completion notification" + (" (held: quiet hours)" if held else ""), ) self.series.append(pt) def step(self, i0: int, i1: int) -> None: """Replay readings[i0:i1] through the detector (a chunk of the cycle).""" if self._aborted or not self.ready: return try: for ts, power in self.readings[i0:i1]: self._watchdog_keepalives(ts) self.cursor["t"] = (ts - self.base).total_seconds() self.detector.process_reading(power, ts) self._note_stall() self._last_real = (ts, power) self._sample(ts) except Exception as exc: # pylint: disable=broad-exception-caught self._aborted = True _LOGGER.debug( "Playground detail replay failed for %s: %s", self.cycle.get("id"), exc ) def _note_stall(self) -> None: """A ``stalled`` event when the detector starts showing a stall (#452). The same display rule live shows (``CycleDetector.stalled``) and the moment the manager fires ``ha_washdata_cycle_stalled``: real readings only. """ stalled = getattr(self.detector, "stalled", False) is True if stalled == getattr(self, "_stall_shown", False): return self._stall_shown = stalled if stalled: info = self.detector.stall_info() or {} self._emit("stalled", f"stalled at {info.get('plateau_w')} W (display only)", "warn") def _watchdog_keepalives(self, until: datetime) -> None: """Inject the keepalives live would have injected before ``until`` (item 390). Mirrors the low-power branch of ``manager._watchdog_check_stuck_cycle``: while the detector waits below the stop threshold and the plug has been silent for more than ``watchdog_interval``, each tick feeds the sensor's last value as an observed synthetic reading. Without it a silent stretch reached the detector as one interval at the next real reading - after the power had already come back - so the end gates were never evaluated inside it, and a soak that live 0.5.7 ends on (item 290 credits the silence) replayed as one cycle. Ticks sit half an interval into each period, the mean phase of a free-running timer. Traces recorded on 0.5.7+ already hold these readings, so for them this is a no-op. The watchdog's staleness force-end and ghost/zombie branches are not emulated. """ last = self._last_real step = self.watchdog_s if last is None or step <= 0: return prev_ts, prev_w = last k = 1 while k <= _MAX_KEEPALIVES_PER_GAP: ts = prev_ts + timedelta(seconds=step * (k + 0.5)) if ts >= until or not self.detector.is_waiting_low_power(): return self.cursor["t"] = (ts - self.base).total_seconds() self.detector.process_reading(prev_w, ts, synthetic=True, observed=True) self._sample(ts) k += 1 def run_tail(self) -> None: """Synthetic quiet tail so a natural end can fire. While the detector holds an envelope-verified pause every ENDING finalize is deferred, and live keeps waiting - its watchdog extends its own silence limit to ``DEFAULT_MAX_DEFERRAL_SECONDS`` + 30 min for the same reason. So the tail runs on for as long as one is held (bounded by that same limit), then the usual span, instead of force-ending a cycle the release rules (95% of the envelope, #375) would have ended a few minutes later. """ if self._aborted or not self.ready: return try: last_ts = self.readings[-1][0] tail_span = _tail_span_s(self.config) step = 30.0 n_steps = min(int(tail_span / step) + 1, 400) max_steps = n_steps + int(_TAIL_VERIFIED_PAUSE_CAP_S / step) budget = n_steps i = 0 while i < budget: i += 1 ts = last_ts + timedelta(seconds=step * i) self.cursor["t"] = (ts - self.base).total_seconds() self.detector.process_reading(0.0, ts) self._sample(ts) if self.detector.state in (STATE_OFF, STATE_FINISHED) and self.captured: break if getattr(self.detector, "_verified_pause", False): budget = min(max_steps, max(budget, i + n_steps)) if not self.captured and self.detector.state != STATE_OFF: flush_ts = last_ts + timedelta(seconds=step * (i + 2)) self.cursor["t"] = (flush_ts - self.base).total_seconds() self.detector.force_end(flush_ts) except Exception as exc: # pylint: disable=broad-exception-caught _LOGGER.debug( "Playground detail replay failed for %s: %s", self.cycle.get("id"), exc ) def finalize(self) -> dict[str, Any]: outcome = self.outcome last_match = dict(self.last_match) # --- outcome --- if self.captured: idx = max( range(len(self.captured)), key=lambda i: float(self.captured[i].get("duration") or 0.0), ) primary = self.captured[idx] outcome["detected"] = True outcome["detected_count"] = len(self.captured) outcome["termination_reason"] = primary.get("termination_reason") outcome["status"] = primary.get("status") outcome["final_duration_s"] = _safe_float(primary.get("duration")) # The program the manager displayed when THAT cycle ended, not whatever # a later sub-cycle left behind. end = self.cycle_ends[idx] if idx < len(self.cycle_ends) else None if end is not None: outcome["end_offset_s"] = _safe_float(end.get("t")) last_match.update( name=end["program"], conf=float(end["confidence"] or 0.0), expected=float(end["expected"] or 0.0), ambiguous=end["ambiguous"], ) label, reason = self._label_decision(primary, end["live_result"]) outcome["would_label"] = label is not None outcome["label_profile"] = label outcome["label_reason"] = reason outcome["matched_profile"] = last_match["name"] outcome["confidence"] = ( round(float(last_match["conf"]), 3) if last_match["name"] else None ) outcome["expected_s"] = ( round(float(last_match["expected"] or 0.0), 1) or None ) if outcome["detected"] and last_match["name"] and self.label: outcome["match_correct"] = last_match["name"].strip() == self.label.strip() # Projected energy/cost: the last LIVE estimate (the post-finish tail resets # the detector's accumulated energy, so series[-1] would read None). for pt in reversed(self.series): if pt.get("projected_energy_wh") is not None: outcome["projected_energy_wh"] = pt.get("projected_energy_wh") outcome["projected_cost"] = pt.get("projected_cost") break # --- finish + milestone markers (reuse production predicates) --- if self.captured and self.finish_configured: # No finished push for an interrupted cycle (audit MANAGER-10); the # milestone below does not depend on it, as live. if notif_rules.cycle_end_is_finish(outcome["status"]): held = self._held(self.cursor["t"]) self._emit( "notify_held" if held else "notify_finish", "finish notification" + (" (held: quiet hours)" if held else ""), ) try: prev_life = int(self.store.get_lifetime_cycle_count()) except Exception: # pylint: disable=broad-exception-caught prev_life = 0 crossed = notif_rules.milestone_crossed( prev_life, prev_life + 1, self.options.get(CONF_NOTIFY_MILESTONES, DEFAULT_NOTIFY_MILESTONES), ) if crossed is not None: # Milestone notifications are held during quiet hours (live contract). m_held = self._held(self.cursor["t"]) self._emit( "notify_held" if m_held else "notify_milestone", f"milestone {crossed} cycles" + (" (held: quiet hours)" if m_held else ""), ) # --- alerts --- alerts: list[dict[str, Any]] = [] expected_dur = float(last_match["expected"] or 0.0) final_dur = outcome["final_duration_s"] or 0.0 if not outcome["detected"]: alerts.append({"code": "did_not_finish", "severity": "error", "detail": "Cycle never reached a terminal state in the replay."}) if outcome["detected"] and (outcome["detected_count"] or 0) > 1: alerts.append({"code": "false_end", "severity": "error", "detail": f"Split into {outcome['detected_count']} cycles."}) if outcome["matched_profile"] is None: alerts.append({"code": "unmatched", "severity": "warn", "detail": "No profile matched this cycle."}) if last_match["ambiguous"]: alerts.append({"code": "ambiguous", "severity": "warn", "detail": "Match was ambiguous (two programs scored close)."}) # How the cycle ended: predictive (smart / terminal-drop) vs the static # low-power fallback. Under auto-detect an unmatched cycle cannot use smart # end-prediction, so it only stops once power stays low for the off-delay - # or, if it never goes quiet, not at all. Surface which happened. term = str(outcome.get("termination_reason") or "") if outcome["detected"] and term == str(TerminationReason.FORCE_STOPPED): alerts.append({"code": "would_run_indefinitely", "severity": "error", "detail": ("The cycle never ended on its own - only the safety " "force-stop finalized it in simulation. In real use it " "would keep counting as running until power stays low.")}) elif outcome["detected"] and term == str(TerminationReason.TIMEOUT): off_min = max(1, round(float(getattr(self.config, "off_delay", 0) or 0) / 60)) if outcome["matched_profile"] is None: alerts.append({"code": "timeout_end", "severity": "warn", "detail": (f"Ended only by the low-power timeout: no profile matched, " f"so smart end-prediction could not run and it waited out " f"the {off_min} min off-delay after power dropped.")}) else: alerts.append({"code": "timeout_end", "severity": "info", "detail": (f"Ended by the low-power timeout, not smart prediction: it " f"waited out the {off_min} min off-delay after power dropped.")}) if expected_dur > 0 and final_dur > 0: ratio = final_dur / expected_dur outcome["overrun_ratio"] = round(ratio, 3) if ratio >= CYCLE_OVERRUN_ANOMALY_RATIO: alerts.append({"code": "overrun", "severity": "warn", "detail": f"Ran {ratio:.0%} of the profile's typical duration."}) elif ratio <= CYCLE_UNDERRUN_ANOMALY_RATIO: alerts.append({"code": "underrun", "severity": "warn", "detail": f"Finished at {ratio:.0%} of typical duration."}) series = self.series if len(series) > MAX_SERIES_PER_CYCLE: # Thin evenly so the shape is preserved (first + last always kept). # Span (len-1)/(N-1) so the final index lands on the true last point # (a plain len/N tops out below it and drops the terminal sample). step = (len(series) - 1) / (MAX_SERIES_PER_CYCLE - 1) series = [series[round(i * step)] for i in range(MAX_SERIES_PER_CYCLE)] return { "cycle_id": self.cycle.get("id"), "label": self.label, "duration_s": self.stored_duration, "start_time": self.cycle.get("start_time"), "active_end_s": _safe_float(self.active_end_s), "active_span_s": _safe_float(self.active_span_s), "config_summary": _sim_config_summary(self.config), "series": series, "events": self.events, "alerts": alerts, "outcome": outcome, } def _sim_config_summary(config: CycleDetectorConfig) -> dict[str, Any]: """Compact view of the effective detector config used for the sim.""" return { "device_type": getattr(config, "device_type", None), "min_power": getattr(config, "min_power", None), "off_delay": getattr(config, "off_delay", None), "min_off_gap": getattr(config, "min_off_gap", None), "start_threshold_w": getattr(config, "start_threshold_w", None), "stop_threshold_w": getattr(config, "stop_threshold_w", None), "anti_wrinkle_enabled": getattr(config, "anti_wrinkle_enabled", None), "anti_wrinkle_max_power": getattr(config, "anti_wrinkle_max_power", None), "anti_wrinkle_max_duration": getattr(config, "anti_wrinkle_max_duration", None), "anti_wrinkle_exit_power": getattr(config, "anti_wrinkle_exit_power", None), "anti_wrinkle_idle_timeout": getattr(config, "anti_wrinkle_idle_timeout", None), "dishwasher_end_spike_quiet_release": getattr(config, "dishwasher_end_spike_quiet_release", None), "smart_termination_duration_ratio": getattr(config, "smart_termination_duration_ratio", None), # Both are clamped by the detector wherever it reads them, so report the # effective figure: a summary carrying an out-of-range override would # describe a sim that did not run. "anti_crease_finalize_ratio": effective_anticrease_finalize_ratio( getattr(config, "anti_crease_finalize_ratio", None) ), "curve_preroll_seconds": effective_curve_preroll_seconds( getattr(config, "curve_preroll_seconds", None) ), } # ─── Test-on-history rows + before/after diff ─────────────────────────────────── # A detected end more than this before the appliance's last activity is an early # end (devtools/end_gate_eval.py's `early_1min`): the cycle was cut short. _EARLY_END_S = 60.0 # A run whose longest piece covers less than this share of its active span did not # survive as one cycle (end_gate_eval's split rule). _SPLIT_SPAN_FRAC = 0.9 def _end_timing(detail: dict[str, Any]) -> tuple[float | None, bool, bool]: """``(end_lag_s, early_end, split)`` of one replay (audit PLAYGROUND-08). The lag is when WashData said "done" minus when the appliance last drew power, the number ``devtools/end_gate_eval.py`` measures; the stored duration is trimmed back to the last activity (item 297) and cannot show it. A run that never finished counts as split: it did not survive as one cycle. """ o = detail.get("outcome", {}) or {} if not o.get("detected"): return None, False, True end_t = o.get("end_offset_s") active_end = detail.get("active_end_s") lag = ( round(float(end_t) - float(active_end), 1) if end_t is not None and active_end is not None else None ) span = detail.get("active_span_s") final = o.get("final_duration_s") split = int(o.get("detected_count") or 0) > 1 or bool( span and final is not None and float(final) < _SPLIT_SPAN_FRAC * float(span) ) return lag, lag is not None and lag < -_EARLY_END_S, split def _detail_to_row(detail: dict[str, Any]) -> dict[str, Any]: """Compact per-cycle row for the Test-on-history table from a detail sim.""" o = detail.get("outcome", {}) end_lag, early_end, split = _end_timing(detail) return { "cycle_id": detail.get("cycle_id"), "label": detail.get("label"), "detected": bool(o.get("detected")), "detected_count": int(o.get("detected_count") or 0), "matched_profile": o.get("matched_profile"), "match_correct": o.get("match_correct"), "confidence": o.get("confidence"), "termination_reason": ( str(o.get("termination_reason")) if o.get("termination_reason") else None ), "status": o.get("status"), "duration_s": o.get("final_duration_s"), "stored_duration_s": detail.get("duration_s"), "expected_s": o.get("expected_s"), "overrun_ratio": o.get("overrun_ratio"), "would_label": bool(o.get("would_label")), "label_profile": o.get("label_profile"), "label_reason": o.get("label_reason"), "alerts": [a.get("code") for a in detail.get("alerts", [])], "end_lag_s": end_lag, "early_end": early_end, "split": split, # "Last N" can reach past the panel's loaded page of cycles, so the row # carries its own date (PLAYGROUND-18). "start_time": (detail.get("start_time") if isinstance(detail.get("start_time"), str) else None), } def _rows_summary(rows: list[dict[str, Any]]) -> dict[str, Any]: total = len(rows) detected = sum(1 for r in rows if r["detected"]) labelled = [r for r in rows if r["label"]] correct = sum(1 for r in labelled if r["match_correct"] is True) wrong = sum(1 for r in labelled if r["match_correct"] is False) unmatched = sum(1 for r in rows if r["detected"] and r["matched_profile"] is None) false_end = sum(1 for r in rows if (r["detected_count"] or 0) > 1) return { "cycles": total, "detected": detected, "labelled": len(labelled), "match_correct": correct, "match_wrong": wrong, "unmatched": unmatched, "false_end": false_end, "early_end": sum(1 for r in rows if r.get("early_end")), "split": sum(1 for r in rows if r.get("split")), } def _run_rows( store: Any, cycles: list[dict[str, Any]], base_config: CycleDetectorConfig, settings_override: dict[str, Any] | None, options: dict[str, Any], price: float | None, prebuilt: tuple[Any, Any, Any, Any] | None = None, ) -> list[dict[str, Any]]: # Snapshots are store-derived (independent of the cycle and the detector-level # settings_override), so build them ONCE and reuse across all cycles/values. # Callers that drive many chunks should pass prebuilt= to avoid rebuilding per chunk. if prebuilt is None: prebuilt = _build_match_snapshots(store) rows: list[dict[str, Any]] = [] for cycle in cycles: detail = simulate_cycle_detail( cycle, base_config, settings_override, store, options, price, compute_series=False, prebuilt=prebuilt, ) if "error" in detail: continue rows.append(_detail_to_row(detail)) return rows def _select_cycles( store: Any, cycle_ids: list[str] | None, count: int | None = None ) -> list[dict[str, Any]]: """The stored cycles a History / Optimize run replays, at most ``MAX_BATCH_CYCLES``: ``cycle_ids`` in order (unknown ids dropped); else the ``count`` most recent, newest first - the panel's "Last N", which its loaded page of 25 cycles could not supply as ids (audit PLAYGROUND-18); else the most recent ``DEFAULT_RECENT_CYCLES`` in stored order.""" past = [c for c in (store.get_past_cycles() or []) if isinstance(c, dict)] if cycle_ids: by_id = {c.get("id"): c for c in past} selected = [by_id[c] for c in cycle_ids if c in by_id] elif count: n = max(1, min(MAX_BATCH_CYCLES, int(count))) selected = list(reversed(past[-n:])) else: selected = past[-DEFAULT_RECENT_CYCLES:] return selected[:MAX_BATCH_CYCLES] def run_playground_history( store: Any, cycle_ids: list[str] | None, base_config: CycleDetectorConfig, settings_override: dict[str, Any] | None, options: dict[str, Any] | None, price: float | None, concurrency: int, prebuilt: tuple[Any, Any, Any, Any] | None = None, ) -> dict[str, Any]: """Per-cycle rows for the Test-on-history table, plus a before/after diff when ``settings_override`` is set. Executor-safe; never raises.""" options = options or {} try: concurrency = max(1, min(MAX_BATCH_CYCLES, int(concurrency))) except (TypeError, ValueError, OverflowError): concurrency = MAX_BATCH_CYCLES try: selected = _select_cycles(store, cycle_ids)[:concurrency] except Exception as exc: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground history: get_past_cycles failed: %s", exc) return {"rows": [], "summary": _rows_summary([])} override = settings_override or None rows = _run_rows(store, selected, base_config, override, options, price, prebuilt) payload: dict[str, Any] = {"rows": rows, "summary": _rows_summary(rows)} if override: base_rows = _run_rows(store, selected, base_config, None, options, price, prebuilt) payload["baseline_rows"] = base_rows payload["baseline_summary"] = _rows_summary(base_rows) payload["diff"] = _diff_rows(base_rows, rows) return payload def finalize_history( rows: list[dict[str, Any]], baseline_rows: list[dict[str, Any]], has_override: bool, ) -> dict[str, Any]: """Assemble the Test-on-history payload from rows collected across chunks (used by the server-side task runner). Reuses the same summary/diff helpers as the one-shot :func:`run_playground_history` so there is one aggregation path.""" payload: dict[str, Any] = {"rows": rows, "summary": _rows_summary(rows)} if has_override and baseline_rows: payload["baseline_rows"] = baseline_rows payload["baseline_summary"] = _rows_summary(baseline_rows) payload["diff"] = _diff_rows(baseline_rows, rows) return payload def _diff_rows( baseline: list[dict[str, Any]], override: list[dict[str, Any]] ) -> dict[str, list[str]]: """Which cycles changed between baseline and override runs (keyed by id).""" base_by = {r["cycle_id"]: r for r in baseline} newly_correct: list[str] = [] regressed: list[str] = [] end_timing_changed: list[str] = [] for r in override: b = base_by.get(r["cycle_id"]) if b is None: continue if b["match_correct"] is not True and r["match_correct"] is True: newly_correct.append(r["cycle_id"]) elif b["match_correct"] is True and r["match_correct"] is not True: regressed.append(r["cycle_id"]) bd, od = b.get("duration_s") or 0.0, r.get("duration_s") or 0.0 if b.get("termination_reason") != r.get("termination_reason") or abs(bd - od) > 60.0: end_timing_changed.append(r["cycle_id"]) return { "newly_correct": newly_correct, "regressed": regressed, "end_timing_changed": end_timing_changed, } # ─── Parameter sweep (1D curve) ──────────────────────────────────────────────────── _SWEEP_OBJECTIVES = ( "match_accuracy", "end_timing_accuracy", "false_end_rate", "median_overrun", "ambiguity_rate", # What the end-gate settings actually move (audit PLAYGROUND-08): none of the # five above changed while off_delay 60 -> 1800 s moved the mean end lag # 19.8 -> 25.4 min, because the stored duration is trimmed to the activity. "end_lag", "early_end_rate", "split_rate", ) # Objectives where a LOWER metric is better (best = minimum), so the sweep picks # the right winner and the panel colours the heatmap consistently. _SWEEP_LOWER_IS_BETTER = frozenset({ "false_end_rate", "median_overrun", "ambiguity_rate", "end_lag", "early_end_rate", "split_rate", }) # Smallest change worth recommending over the current value. A rate moves in # steps of one cycle (1/N, passed by the caller), so anything less is the # denominator moving, not a cycle getting better; the end lag is measured on a # replay whose tail steps are 30 s; the overrun deviation is a share of the # profile's length. _SWEEP_MIN_GAIN = {"end_lag": 60.0, "median_overrun": 0.01} def _sweep_is_better(candidate: float, best: float, objective: str) -> bool: if objective in _SWEEP_LOWER_IS_BETTER: return candidate < best return candidate > best def _sweep_min_gain(objective: str, n_cycles: int) -> float: if objective in _SWEEP_MIN_GAIN: return _SWEEP_MIN_GAIN[objective] # A tiny epsilon below one cycle so 1/N itself, after rounding, still counts. return (1.0 / n_cycles - 1e-6) if n_cycles > 0 else 0.0 def _sweep_regresses(summary: Any, baseline: Any) -> bool: """Does ``summary`` lose what the current setting has? The hard guard: a value that ends more cycles early, splits more of them, or detects fewer can never be recommended, whatever its objective says (audit PLAYGROUND-08).""" if not isinstance(summary, dict) or not isinstance(baseline, dict): return False return ( int(summary.get("early_end") or 0) > int(baseline.get("early_end") or 0) or int(summary.get("split") or 0) > int(baseline.get("split") or 0) or int(summary.get("detected") or 0) < int(baseline.get("detected") or 0) ) def sweep_baseline( store: Any, cycle_ids: list[str] | None, base_config: CycleDetectorConfig, objective: str, options: dict[str, Any] | None, price: float | None, prebuilt: tuple[Any, Any, Any, Any] | None = None, ) -> dict[str, Any]: """The current settings' metric and summary over the sweep's cycles: what every swept value has to beat. Executor-safe; never raises.""" try: selected = _select_cycles(store, cycle_ids) rows = _run_rows(store, selected, base_config, None, options or {}, price, prebuilt) metric = objective_metric(rows, objective) return { "metric": round(metric, 4) if metric is not None else None, "summary": _rows_summary(rows), } except Exception as exc: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground sweep baseline failed: %s", exc) return {"metric": None, "summary": None} def finalize_sweep_1d( param: str, objective: str, points: list[dict[str, Any]], current_value: Any, baseline: dict[str, Any] | None = None, ) -> dict[str, Any]: """Assemble a 1D sweep payload from per-value points collected across chunks. With a ``baseline`` (:func:`sweep_baseline`, the current settings on the same cycles) the recommendation is conservative (audit PLAYGROUND-08): a value that regresses early ends, splits or detection is never best; and unless the best value beats the current one by at least one cycle (:func:`_sweep_min_gain`) the answer is to keep the current value - a tie used to pick the first swept value, so "Apply best" offered ``off_delay=60``. Without one, ties still prefer the current value over the first.""" base_summary = (baseline or {}).get("summary") base_metric = (baseline or {}).get("metric") n_cycles = int((base_summary or {}).get("cycles") or 0) or max( (int((p.get("summary") or {}).get("cycles") or 0) for p in points), default=0 ) def _is_current(value: Any) -> bool: try: return current_value is not None and abs(float(value) - float(current_value)) < 1e-6 except (TypeError, ValueError, OverflowError): return False best: dict[str, Any] | None = None for p in points: m = p.get("metric") guarded = base_summary is not None and _sweep_regresses(p.get("summary"), base_summary) p["guarded"] = guarded if m is None or guarded: continue if ( best is None or _sweep_is_better(m, best["metric"], objective) or (m == best["metric"] and _is_current(p["value"])) ): best = {"value": p["value"], "metric": m} keep_current = False if base_metric is not None: gain = _sweep_min_gain(objective, n_cycles) if best is None or not ( (base_metric - best["metric"] if objective in _SWEEP_LOWER_IS_BETTER else best["metric"] - base_metric) >= gain ): best = {"value": current_value, "metric": base_metric} keep_current = True elif best is not None and _is_current(best["value"]): keep_current = True return { "param": param, "objective": objective, "points": points, "current_value": current_value, "current_metric": base_metric, "current_summary": base_summary, "best_value": best["value"] if best else None, "best_metric": best["metric"] if best else None, "keep_current": keep_current, "cycles": n_cycles, "lower_is_better": objective in _SWEEP_LOWER_IS_BETTER, } def objective_metric(rows: list[dict[str, Any]], objective: str) -> float | None: """Reduce a set of per-cycle rows to a single objective metric (0-1, or a ratio for median_overrun). Higher is better EXCEPT false_end_rate / median_overrun deviation (the caller/panel knows the direction).""" if not rows: return None detected = [r for r in rows if r["detected"]] labelled = [r for r in detected if r["label"]] if objective == "match_accuracy": if not labelled: return None return sum(1 for r in labelled if r["match_correct"] is True) / len(labelled) if objective == "false_end_rate": if not detected: return None return sum(1 for r in detected if (r["detected_count"] or 0) > 1) / len(detected) if objective == "ambiguity_rate": if not detected: return None return sum(1 for r in detected if "ambiguous" in (r["alerts"] or [])) / len(detected) if objective == "end_timing_accuracy": # Fraction of cycles whose *detected* end lands within 10% of that cycle's own # recorded duration (its true end) - NOT the profile median. Scoring against # the median would reward ending at the typical length even for a cycle that # legitimately ran long or short, so the sweep must compare to stored_duration_s. ok = 0 n = 0 for r in detected: ref = float(r.get("stored_duration_s") or 0.0) dur = float(r.get("duration_s") or 0.0) if ref <= 0 or dur <= 0: continue n += 1 if abs(dur - ref) <= 0.10 * ref: ok += 1 return (ok / n) if n else None if objective == "end_lag": # Median seconds from the appliance's last activity to WashData's "done". lags = [float(r["end_lag_s"]) for r in detected if r.get("end_lag_s") is not None] if not lags: return None lags.sort() mid = len(lags) // 2 return lags[mid] if len(lags) % 2 else (lags[mid - 1] + lags[mid]) / 2.0 if objective in ("early_end_rate", "split_rate"): # Over EVERY replayed cycle, not just the detected ones: a value that stops # detecting a hard cycle must not shrink its own denominator and win. key = "early_end" if objective == "early_end_rate" else "split" return sum(1 for r in rows if r.get(key)) / len(rows) if objective == "median_overrun": # Score by the median duration's DEVIATION from the profile's expected # duration (|ratio - 1|), so "best" is the value that makes cycles land # closest to their typical length - not the smallest raw ratio (which # would reward a severe *underrun*, e.g. 0.5x, as if it were ideal). ratios = sorted( float(r["overrun_ratio"]) for r in detected if r.get("overrun_ratio") ) if not ratios: return None mid = len(ratios) // 2 median = ratios[mid] if len(ratios) % 2 else (ratios[mid - 1] + ratios[mid]) / 2.0 return abs(median - 1.0) return None def _coerce_param(base_config: CycleDetectorConfig, param: str, value: float) -> Any: """Coerce a sweep value to the override map's expected type.""" mapping = _OVERRIDE_FIELD_MAP.get(param) if mapping is None: return value _field, coerce = mapping try: return coerce(value) except (TypeError, ValueError, OverflowError): return value def run_playground_sweep( store: Any, cycle_ids: list[str] | None, base_config: CycleDetectorConfig, param: str, values: list[float], objective: str, options: dict[str, Any] | None, price: float | None, concurrency: int, prebuilt: tuple[Any, Any, Any, Any] | None = None, ) -> dict[str, Any]: """Sweep one param and score each value by ``objective`` computed from the per-cycle rows. Executor-safe; never raises. (The 2D heatmap was removed: the panel never sent a second parameter, audit PLAYGROUND.) """ options = options or {} if objective not in _SWEEP_OBJECTIVES: objective = "match_accuracy" try: concurrency = max(1, min(MAX_BATCH_CYCLES, int(concurrency))) except (TypeError, ValueError, OverflowError): concurrency = MAX_BATCH_CYCLES try: selected = _select_cycles(store, cycle_ids)[:concurrency] except Exception as exc: # pylint: disable=broad-exception-caught _LOGGER.debug("Playground sweep: get_past_cycles failed: %s", exc) return {"error": "no cycles"} def _metric_for(override: dict[str, Any]) -> tuple[float | None, dict[str, Any]]: rows = _run_rows(store, selected, base_config, override, options, price, prebuilt) return objective_metric(rows, objective), _rows_summary(rows) current_x = _sim_config_summary(base_config).get( _OVERRIDE_FIELD_MAP.get(param, (param,))[0] ) if current_x is None: # The summary carries only part of the config, so 8 sweepable keys (the # completion/interrupted/start-duration thresholds, the two ratios, ...) # reported no current value and the panel drew no marker for it. try: current_x = effective_settings( base_config, prebuilt[1] if prebuilt else None ).get(param) except Exception: # pylint: disable=broad-exception-caught current_x = None points: list[dict[str, Any]] = [] for vx in values: override = {param: _coerce_param(base_config, param, vx)} metric, summary = _metric_for(override) points.append( {"value": vx, "metric": round(metric, 4) if metric is not None else None, "summary": summary} ) # One selection rule for the one-shot and the chunked task (which adds the # current-settings baseline the guard and the keep-current rule need). return finalize_sweep_1d(param, objective, points, current_x) # ─── DTW debug ──────────────────────────────────────────────────────────────── def _safe_float(value: Any) -> float | None: try: if value is None: return None return round(float(value), 2) except (TypeError, ValueError, OverflowError): return None