5993 lines
311 KiB
Python
5993 lines
311 KiB
Python
# WashData - Home Assistant integration for appliance cycle monitoring via smart plugs.
|
|
# Copyright (C) 2026 Lukas Bandura
|
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
#
|
|
# This program is free software: you can redistribute it and/or modify
|
|
# it under the terms of the GNU Affero General Public License as published
|
|
# by the Free Software Foundation, either version 3 of the License, or
|
|
# (at your option) any later version.
|
|
#
|
|
# This program is distributed in the hope that it will be useful,
|
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
# GNU Affero General Public License for more details.
|
|
#
|
|
# You should have received a copy of the GNU Affero General Public License
|
|
# along with this program. If not, see <https://www.gnu.org/licenses/>.
|
|
"""Cycle detection logic for WashData."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import itertools
|
|
import logging
|
|
import math
|
|
import traceback
|
|
from dataclasses import dataclass
|
|
from datetime import datetime, timedelta
|
|
from typing import Any, Callable, cast
|
|
import numpy as np
|
|
|
|
from homeassistant.util import dt as dt_util
|
|
|
|
from . import match_rules
|
|
from .log_utils import DeviceLoggerAdapter
|
|
from .time_utils import utc_now
|
|
from .const import (
|
|
TERMINAL_EVENT_PEAK_FRAC,
|
|
DISHWASHER_QUIET_RELEASE_TERMINAL_MARGIN,
|
|
ANTI_WRINKLE_ELIGIBLE_REASONS,
|
|
TerminationReason,
|
|
STATE_OFF,
|
|
STATE_DELAY_WAIT,
|
|
STATE_STARTING,
|
|
STATE_RUNNING,
|
|
STATE_PAUSED,
|
|
STATE_ENDING,
|
|
STATE_FINISHED,
|
|
STATE_ANTI_WRINKLE,
|
|
STATE_INTERRUPTED,
|
|
STATE_FORCE_STOPPED,
|
|
STATE_UNKNOWN,
|
|
STATE_IDLE,
|
|
DEVICE_TYPE_WASHING_MACHINE,
|
|
DEVICE_TYPE_DRYER,
|
|
DEVICE_TYPE_WASHER_DRYER,
|
|
DEFAULT_MAX_DEFERRAL_SECONDS,
|
|
DEFAULT_DEFER_FINISH_CONFIDENCE,
|
|
DEFAULT_SMART_TERMINATION_DURATION_RATIO,
|
|
DISHWASHER_END_SPIKE_MIN_PROGRESS,
|
|
DISHWASHER_END_SPIKE_QUIET_RELEASE_SECONDS,
|
|
DISHWASHER_END_SPIKE_WAIT_SECONDS,
|
|
DISHWASHER_SMART_TERMINATION_DEBOUNCE_SECONDS,
|
|
SMART_TERM_TAIL_MAX_RATIO,
|
|
SMART_TERM_TAIL_MIN_POINTS,
|
|
SMART_TERM_TAIL_WINDOW_FRAC,
|
|
SMART_TERM_TAIL_WINDOW_MIN_S,
|
|
SMART_TERM_TAIL_WINDOW_S,
|
|
WASHER_SMART_TERMINATION_DEBOUNCE_MAX_SECONDS,
|
|
STARTING_PAUSED_TRUE_OFF_TIMEOUT_SECONDS,
|
|
DISHWASHER_MATCH_FREEZE_QUIET_SECONDS,
|
|
DISHWASHER_MIN_CYCLE_DURATION_S,
|
|
TERMINAL_DROP_OFF_DELAY_SECONDS,
|
|
ENDING_HARD_FINALIZE_RATIO,
|
|
ENDING_HARD_FINALIZE_MIN_QUIET_S,
|
|
GATE_CADENCE_MEDIAN_FACTOR,
|
|
END_GATE_LATE_RATIO,
|
|
resolve_end_gate_late_ratio,
|
|
END_GATE_LATE_SECONDS,
|
|
END_GATE_HAZARD_MARGIN,
|
|
END_GATE_HAZARD_MIN_CYCLES,
|
|
END_GATE_HAZARD_POSITION_SLACK,
|
|
STANDBY_BAND_FINALIZE_DEVICE_TYPES,
|
|
STANDBY_BAND_MIN_RATIO,
|
|
STANDBY_BAND_LOOSE_MIN_RATIO,
|
|
STANDBY_BAND_NEAR_STOP_FACTOR,
|
|
STANDBY_BAND_NEAR_STOP_W,
|
|
DEVICE_TYPE_DISHWASHER,
|
|
TERMINAL_QUIET_CAP_S,
|
|
TRUSTED_LENGTH_FLOOR_FRAC,
|
|
STANDBY_BAND_WINDOW_S,
|
|
STANDBY_BAND_MAX_FRACTION,
|
|
STANDBY_BAND_FLATNESS_FRACTION,
|
|
STANDBY_BAND_FLATNESS_FLOOR_W,
|
|
DEFAULT_ANTI_CREASE_FINALIZE_RATIO,
|
|
ANTI_CREASE_FINALIZE_RATIO_MIN,
|
|
ANTI_CREASE_FINALIZE_RATIO_MAX,
|
|
DEFAULT_CURVE_PREROLL_SECONDS,
|
|
CURVE_PREROLL_MAX_SECONDS,
|
|
PREROLL_CHAIN_BREAK_SECONDS,
|
|
ANTI_CREASE_CONFIRM_WINDOW_S,
|
|
ANTI_CREASE_TERMINAL_HIGH_MIN_FRAC,
|
|
ANTI_CREASE_TERMINAL_MATCH_FRAC,
|
|
ANTI_CREASE_SPIN_WAIT_MAX_RATIO,
|
|
)
|
|
|
|
# The dishwasher end-spike wait window is shared between two code paths
|
|
# (Smart Termination's wait branch and _should_defer_finish's no-end-spike
|
|
# branch). They MUST release the cycle at the same instant - sanity-check
|
|
# that the constants module loaded a sensible value rather than allowing the
|
|
# paths to silently drift if one was changed and the other forgotten.
|
|
if DISHWASHER_END_SPIKE_WAIT_SECONDS <= 0:
|
|
# Runtime check (not assert: asserts are stripped under python -O).
|
|
raise ValueError("DISHWASHER_END_SPIKE_WAIT_SECONDS must be positive")
|
|
|
|
# Opt-in ML end-detection guard (Stage 6). When the manager injects an
|
|
# end-confidence provider (only when the user enabled ML models for the device),
|
|
# the cycle-end model can defer a *normal* completion if it judges the current
|
|
# low-power event to be a pause rather than the true end. This is intentionally
|
|
# asymmetric: it can only *delay* a completion, never end a cycle early, and it
|
|
# is bounded, so a wrong model can slow a finish but can neither stop one early
|
|
# nor hang the cycle. Force-stop / smart-termination / user paths never consult
|
|
# it. Overridable emphasis lives here rather than const.py to keep the guard
|
|
# self-contained (it is detector-internal policy, not user configuration).
|
|
ML_END_GUARD_MIN_CONFIDENCE = 0.5 # P(true end) below this -> treat as a likely pause
|
|
ML_END_GUARD_MAX_DEFER_SECONDS = 1800.0 # cap the extra wait the guard may add (30 min)
|
|
# The opt-in ML end-guard / terminal-drop providers rebuild the trace and run
|
|
# inference on every ENDING-phase evaluation. During a long quiet tail (e.g. a
|
|
# dishwasher's up-to-1h soak) that is wasteful, so recompute at most this often
|
|
# (data-clock seconds). Safe to cache: the guard only ever *defers* and terminal
|
|
# drop only ever *shortens*, so both tolerate a value up to this window stale.
|
|
ML_PROVIDER_THROTTLE_SECONDS = 30.0
|
|
# ENDING energy gate vs a flat standby (audit DETECT-08, item 95 residual). The
|
|
# gate's bar is end_energy_threshold x 3600 / off_delay as a MEAN power (1.0 W at
|
|
# 180 s, 0.1 W at 1800 s), so a standby sitting between that bar and the stop
|
|
# threshold pinned it forever: an unmatched cycle ran to the 8 h cap. A window is
|
|
# a flat standby when every reading of the last max(off_delay,
|
|
# STANDBY_BAND_WINDOW_S) seconds is above 0 W and the window spans no more than
|
|
# max(FLOOR_W, FRAC x its highest reading). The non-zero floor is what keeps the
|
|
# gate's one measured catch: a dishwasher's passive drying flickering 0-0.5 W
|
|
# for 25 min before its pump-out (Hatton ECO `44d35b4ca01e`).
|
|
#
|
|
# Two tiers, like the standby band's. Just under stop (every reading >=
|
|
# NEAR_STOP_FRAC x stop_threshold_w) is the item-95 shape and ends after the
|
|
# un-shortened max(off_delay, min_off_gap). Any other flat window waits
|
|
# LOOSE_QUIET_S: flat non-zero phases far under stop DO resume in real cycles.
|
|
# Over every stored trace in the corpus (all formats) the resumed flat non-zero
|
|
# sub-stop pauses longer than that device's min_off_gap are two, and both sit at
|
|
# about half of stop: an 18.9 min washer soak at 2.7-3.0 W on a 6 W stop (it split
|
|
# with one tier at min_off_gap, end_gate_eval --shipped-watchdog) and an 80.8 min
|
|
# dishwasher drying phase at 0.7-0.9 W on a 1.5 W stop (it split replayed
|
|
# unmatched). Neither reaches 0.75 x stop; 2 h is 1.5x the longer one.
|
|
ENDING_FLAT_STANDBY_MIN_READINGS = 3
|
|
ENDING_FLAT_STANDBY_SPREAD_FRAC = 0.25
|
|
ENDING_FLAT_STANDBY_SPREAD_FLOOR_W = 0.2
|
|
ENDING_FLAT_STANDBY_NEAR_STOP_FRAC = 0.75
|
|
ENDING_FLAT_STANDBY_LOOSE_QUIET_S = 7200.0
|
|
if not 0 < DISHWASHER_END_SPIKE_MIN_PROGRESS < 1:
|
|
raise ValueError("DISHWASHER_END_SPIKE_MIN_PROGRESS must be a fraction in (0, 1)")
|
|
from .signal_processing import (
|
|
energy_gap_threshold_s,
|
|
integrate_wh,
|
|
median_fast,
|
|
percentile_linear,
|
|
terminal_event_end,
|
|
terminal_quiet_seen,
|
|
)
|
|
|
|
_LOGGER = logging.getLogger(__name__)
|
|
|
|
# After a user/external stop the manual-stop lockout swallows the machine's
|
|
# spin-down/drain so it is not logged as a fresh cycle. The lockout normally
|
|
# clears as soon as power drops to idle. As a safety net, if power instead stays
|
|
# at or above the start threshold for longer than any plausible spin-down, the
|
|
# device is running a genuinely new (back-to-back) load: release the lockout so
|
|
# the new cycle is detected instead of being pinned until the progress-reset
|
|
# window expires (issue #267).
|
|
STOP_LOCKOUT_RELEASE_SECONDS = 180.0
|
|
|
|
# Register item 501: how much of start_energy_threshold a standby RE-probe (one
|
|
# that began with no reading below stop_threshold_w since the last false start)
|
|
# must fill before entities show it as `starting`. Display only: detection never
|
|
# reads it. Measured by devtools/start_gate_eval.py (flickers per idle day).
|
|
STANDBY_REPROBE_SHOW_ENERGY_FRACTION = 0.5
|
|
|
|
# Register item 515: a false start out of Finished / Interrupted / Force-Stopped
|
|
# returns to that state with its original entry time (as item 504 returns one
|
|
# out of DELAY_WAIT), instead of falling to OFF. With it the manager clears the
|
|
# cycle end, Clean and the unload nag only when a probe commits (RUNNING), not
|
|
# when it begins: a probe that aborted used to lose all three, and a straddling
|
|
# standby (#35) probes on almost every reading. The A/B switch for the revert
|
|
# checks and `devtools/start_gate_eval.py` (`terminal_lost`).
|
|
TERMINAL_PROBE_RETURNS = True
|
|
|
|
# Discussion #452: two more display-only states, read by `exposed_state` and never
|
|
# by detection.
|
|
#
|
|
# STALLED. A washer that halts mid-cycle (an unbalanced load, a door warning)
|
|
# sits at its standby draw, just ABOVE stop_threshold_w, so the cycle never even
|
|
# pauses. A stall is a flat run of readings in the standby band's near-stop shape
|
|
# (`standby_near_stop_ceiling`, at most STANDBY_BAND_MAX_FRACTION of the cycle's
|
|
# peak, spread within the band's flatness limit) that has lasted longer than any
|
|
# low stretch the matched programme recorded from that position on
|
|
# (END_GATE_HAZARD_MARGIN x the longest in its near-stop pause catalogue, element
|
|
# 15; at least STALL_MIN_S) while the programme still owes work: its terminal
|
|
# high-power block not yet seen (#399's evidence), or, for a programme without
|
|
# one, the run began before STALL_OWES_MAX_POSITION of its expected duration.
|
|
# The evidence is the match the run began under (a plateau soon reads to the
|
|
# matcher as a finished SHORTER programme) AND the current match (the last tick
|
|
# before a real end can be a prefix match of a LONGER one); with no current match
|
|
# the first stands. Unmatched (or matched with fewer than
|
|
# END_GATE_HAZARD_MIN_CYCLES traced cycles): STALL_UNMATCHED_MIN_S. Shown as
|
|
# `paused` with sub-state STALL_SUB_STATE, cleared by the first reading out of
|
|
# the band. While shown it reaches detection twice: the anti-crease finalize
|
|
# waits (item 511, below), and in the standby-band finalize,
|
|
# during a run whose two matches both still owe their terminal high-power block,
|
|
# the near-stop tier is held, so the plateau waits for the loose tier
|
|
# (STANDBY_BAND_LOOSE_MIN_RATIO x expected) instead of closing a halted wash as
|
|
# finished at 1.0x (STALL_HOLDS_STANDBY_BAND, the A/B switch for
|
|
# devtools/end_gate_eval.py).
|
|
STALL_DEVICE_TYPES = STANDBY_BAND_FINALIZE_DEVICE_TYPES
|
|
STALL_MIN_S = 600.0
|
|
STALL_UNMATCHED_MIN_S = 1800.0
|
|
STALL_OWES_MAX_POSITION = 0.75
|
|
STALL_SUB_STATE = "Stalled"
|
|
STALL_HOLDS_STANDBY_BAND = True
|
|
# Register item 511: a halted programme does not advance. Once a stall is over
|
|
# (the wash resumed, or the plug dropped below stop), every duration-based end
|
|
# gate (Smart Termination, the hard finalize, the item-306 shortening, the hazard
|
|
# gate, the minimum-duration deferral, both standby-band tiers, the anti-crease
|
|
# gate and its #399 spin wait) reads the elapsed time LESS that run, from its
|
|
# first reading (`_gate_elapsed_s`); the live matcher reads the trace with it cut
|
|
# out (`_match_readings`) and the #399 scan offset moves past it. Without that the
|
|
# halt carried the clock past each gate's ratio and the resumed wash ended at its
|
|
# next quiet. A stall still shown counts as before, so the standby band closes
|
|
# a display left on after a real end exactly as the #452 hold above allows (the
|
|
# loose tier bounds it), and the anti-crease finalize waits while one is shown. The
|
|
# stored duration, the displayed elapsed and the 8 h cap keep the wall clock.
|
|
# Excluding the stall in progress as well (the near-stop tier then waited out
|
|
# every shown stall) kept 26 more synthetic 45 min halts whole, but held 10 real
|
|
# ends with the display left on for 45 min until it went off (up to 57 min late)
|
|
# and one left on for twice the programme up to 6 h. The A/B switch for the
|
|
# revert checks.
|
|
STALL_EXCLUDED_FROM_GATES = True
|
|
# Register item 514: progress reads programme time too. Every finished stall
|
|
# leaves the elapsed time and the trace the progress estimate reads
|
|
# (`progress_elapsed_s` / `progress_trace`, the one input of the manager and the
|
|
# Playground replay), as the user pause leaves the elapsed time
|
|
# (`manager.net_elapsed_seconds`), and the finished stalls ride in the cycle data
|
|
# (`stall_spans`) so the cycle-end label match reads the trace without them
|
|
# (`match_rules.final_match_input`). The stall shown now as well
|
|
# (STALL_CURRENT_EXCLUDED_FROM_PROGRESS): from the moment it shows, the remaining
|
|
# time stops counting down. Measured (`eta_eval --halt-at 0.5 --halt-min 45`, 246
|
|
# washer targets): through the halt the ETA fell a median 40.9 min before, 0.5 now;
|
|
# its median error 1 / 10 / 30 min after the resume 41.8 / 36.4 / 21.6 -> 13.2 /
|
|
# 16.7 / 13.7 min (finished stalls only: 25.0 / 16.7 / 13.7); real corpus 0 of 450
|
|
# rows changed. A/B switches for the revert checks.
|
|
STALL_EXCLUDED_FROM_PROGRESS = True
|
|
STALL_CURRENT_EXCLUDED_FROM_PROGRESS = True
|
|
# ...and a finished user pause is programme time stood still too (item 514): the
|
|
# gate clock, the #399 scan offset and the live matcher's trace leave it out like
|
|
# a finished stall (the union of the two, `_gate_spans`), and it rides in the
|
|
# cycle data with them (`halt_spans`). The user pause blocked every finisher
|
|
# while it lasted (the verified pause) but the gates read the raw clock after
|
|
# the resume, so a pause that cuts the plug's power ended the resumed wash at
|
|
# its next quiet. Progress already leaves it out (`manager.net_elapsed_seconds`).
|
|
# Measured (`end_gate_eval --halt-at F --halt-min 45 --user-pause`, 258 cycles):
|
|
# power cut at 50% / 30% / 90%, splits 48 / 50 / 40 -> 12 / 22 / 33, early ends
|
|
# 6 / 5 / 18 -> 5 / 7 / 8 (each new one the unpaused cycle's own end, or a former
|
|
# split); paused on the display level, 9 -> 6 and 14 -> 9 (50% / 90%).
|
|
USER_PAUSE_EXCLUDED_FROM_GATES = True
|
|
# Register item 514: a halt is abrupt. A programme winds down to its end (the last
|
|
# spin's ramp, then quiet), so the reading before a display left on is low; a halt
|
|
# stops the machine where it is. A flat run whose previous reading was at least
|
|
# STALL_ABRUPT_PEAK_FRACTION of the cycle's peak (and twice the band's ceiling)
|
|
# began straight out of activity (`_stall_run_abrupt`), and when the MATCHED
|
|
# programme it began under still owes work, the current match need not agree: the
|
|
# plateau soon reads to the matcher as a finished shorter programme, which vetoed
|
|
# the display and the near-stop hold above and closed the halted wash 10 min in.
|
|
# Unmatched runs are left to the current match (a short programme's real end looks
|
|
# the same). Measured (`end_gate_eval --halt-at/--halt-min`, 258 washer,
|
|
# washer-dryer and dryer cycles): at 0.10, 45 min halts at 50% shown 196 -> 198,
|
|
# closed inside the halt 75 -> 67, splits 79 -> 71 (30%: 76 -> 61; 90%: 130 -> 118),
|
|
# early ends unchanged; a display left on after a real end (100%, 45 min and 2x):
|
|
# no new stall flag, 2 closes later, both in runs the old gates had split. At 0.05:
|
|
# 50% splits 79 -> 54 and 9 more shown, but one more display-on flag and one more
|
|
# real end held (+36 min). 0 disables (the A/B switch).
|
|
STALL_ABRUPT_PEAK_FRACTION = 0.10
|
|
#
|
|
# IDLE. Outside a cycle a two-level appliance (display on vs switched off) shows
|
|
# `idle` at its standby level and `off` below it. The level is the user's
|
|
# power-off threshold (#284) when one is set: idle at or above it, off below it,
|
|
# debounced by power_off_delay like the #284 reset itself. Otherwise the learned
|
|
# standby level (`learned_standby_level_w`, set by the manager): idle from
|
|
# IDLE_OFF_FRACTION of it, off below IDLE_EXIT_FRACTION of that (hysteresis), each
|
|
# held IDLE_DEBOUNCE_S, and only once the appliance has been seen below that off
|
|
# level outside a cycle since start-up. A level under IDLE_MIN_STANDBY_W, or at
|
|
# or above the start threshold, is not clearly separated from off or from a
|
|
# start: no idle at all, exactly as before.
|
|
IDLE_MIN_STANDBY_W = 2.0
|
|
IDLE_OFF_FRACTION = 0.5
|
|
IDLE_EXIT_FRACTION = 0.75
|
|
IDLE_DEBOUNCE_S = 60.0
|
|
|
|
|
|
def effective_anticrease_finalize_ratio(value: Any) -> float:
|
|
"""The ratio the anti-crease gate will actually use for a stored value.
|
|
|
|
Only ``ws_set_options`` range-checks this option: ``import_config`` strips
|
|
nulls only, a selective import writes numbers through, and the Playground
|
|
sanitizer just casts to float. So the value is held to its documented range
|
|
here, at every point of use, and anything unusable falls back to the default
|
|
rather than disarming the gate (a stored ``0.0`` would satisfy the
|
|
past-expected test for every duration).
|
|
|
|
Shared with the Playground's config summary so the figure the panel shows and
|
|
the figure the gate applies cannot drift apart.
|
|
"""
|
|
try:
|
|
ratio = float(value)
|
|
except (TypeError, ValueError, OverflowError):
|
|
return DEFAULT_ANTI_CREASE_FINALIZE_RATIO
|
|
if not math.isfinite(ratio):
|
|
return DEFAULT_ANTI_CREASE_FINALIZE_RATIO
|
|
return min(
|
|
ANTI_CREASE_FINALIZE_RATIO_MAX, max(ANTI_CREASE_FINALIZE_RATIO_MIN, ratio)
|
|
)
|
|
|
|
|
|
def effective_curve_preroll_seconds(value: Any) -> float:
|
|
"""The pre-roll window actually applied for a stored value (0 = off).
|
|
|
|
Same reasoning as :func:`effective_anticrease_finalize_ratio`: the detector
|
|
caps the window at ``CURVE_PREROLL_MAX_SECONDS`` wherever it reads it, so the
|
|
summary has to report the capped figure or it describes a sim that did not
|
|
run.
|
|
"""
|
|
try:
|
|
window = float(value or 0.0)
|
|
except (TypeError, ValueError, OverflowError):
|
|
return 0.0
|
|
if not math.isfinite(window) or window <= 0:
|
|
return 0.0
|
|
return min(window, CURVE_PREROLL_MAX_SECONDS)
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class MatchContext:
|
|
"""One live match as the detector consumes it (audit DETECT-15).
|
|
|
|
Producers build this by name; ``update_match`` still reads the positional
|
|
sequence it always did (``as_sequence``), so a field can no longer be put in
|
|
the wrong slot or left out of one producer. Every field after the first five
|
|
defaults to "no evidence", exactly what a shorter legacy tuple meant.
|
|
"""
|
|
|
|
profile_name: str | None
|
|
confidence: float
|
|
expected_duration: float
|
|
phase_name: str | None = None
|
|
is_confident_mismatch: bool = False
|
|
is_ambiguous: bool = False
|
|
# Element 7 is retired: it carried the #364 prefix-fit flag, removed in 0.5.8
|
|
# (see const.py). The slot stays (always False) so 8-14 keep their positions.
|
|
# Element 8: the #288 full-shape term (anti-crease finalize only).
|
|
is_prefix_ambiguous_full_shape: bool = False
|
|
tail_power: Any = None # 9 (#364 power guard)
|
|
terminal_high: Any = None # 10 (#399 anti-crease spin look-ahead)
|
|
terminal_quiet_s: Any = None # 11 (item 297 keep-tail cap)
|
|
longest_candidate_s: float = 0.0 # 12 (item 330 fallback bar)
|
|
trusted_min_s: Any = None # 13 (item 384 trusted-length floor)
|
|
pause_catalogue: Any = None # 14 (DETECT-16 hazard gate)
|
|
# 15 (#452 stall display): the same catalogue below `standby_near_stop_ceiling`,
|
|
# or a zero-argument callable returning it: read only when a flat run lasts
|
|
# long enough to be judged, so the profile's traces are not walked for it on
|
|
# every cycle (`test_perf_budgets`).
|
|
stall_catalogue: Any = None
|
|
|
|
def __getitem__(self, index: Any) -> Any:
|
|
return self.as_sequence()[index]
|
|
|
|
def __len__(self) -> int:
|
|
return 15
|
|
|
|
def as_sequence(self) -> tuple[Any, ...]:
|
|
return (
|
|
self.profile_name, self.confidence, self.expected_duration, self.phase_name,
|
|
self.is_confident_mismatch, self.is_ambiguous, False,
|
|
self.is_prefix_ambiguous_full_shape, self.tail_power, self.terminal_high,
|
|
self.terminal_quiet_s, self.longest_candidate_s, self.trusted_min_s,
|
|
self.pause_catalogue, self.stall_catalogue,
|
|
)
|
|
|
|
|
|
@dataclass
|
|
class CycleDetectorConfig:
|
|
"""Configuration for cycle detection."""
|
|
|
|
min_power: float
|
|
off_delay: int
|
|
device_type: str = DEVICE_TYPE_WASHING_MACHINE
|
|
interrupted_min_seconds: int = 150
|
|
completion_min_seconds: int = 600
|
|
start_duration_threshold: float = 5.0
|
|
start_energy_threshold: float = 0.005
|
|
end_energy_threshold: float = 0.05 # 0.05 Wh (50 mWh) threshold for "still active"
|
|
min_off_gap: int = 60
|
|
start_threshold_w: float = 2.0
|
|
stop_threshold_w: float = 2.0
|
|
# The FINISH-DEFERRAL ratio (`_should_defer_finish`), despite the name: not the
|
|
# matcher's Stage-1 `profile_match_min_duration_ratio`. Built from
|
|
# const.DEFAULT_DEFER_FINISH_RATIO (audit DETECT-02).
|
|
min_duration_ratio: float = 0.8
|
|
# Minimum live-match confidence for a match to be trusted by Smart Termination
|
|
# and the anti-crease gate. Fed from the `profile_match_threshold` option, which
|
|
# up to 0.5.5 was stored and never read - so raising it (the workaround the #288
|
|
# reporter documented) silently did nothing. Default matches the value that was
|
|
# hard-coded at those two sites, so behaviour is unchanged unless the user has
|
|
# deliberately tuned the option.
|
|
match_confidence_threshold: float = 0.4
|
|
# Power-based Off detection (issue #284). Carried on the config so the manager
|
|
# (the single owner of the terminal -> Off transition) can read them live; the
|
|
# detector itself does not act on them. 0 = disabled.
|
|
power_off_threshold_w: float = 0.0
|
|
power_off_delay: float = 30.0
|
|
match_interval: int = 300 # Default profile match interval
|
|
anti_wrinkle_enabled: bool = False
|
|
anti_wrinkle_max_power: float = 400.0
|
|
anti_wrinkle_max_duration: float = 60.0
|
|
anti_wrinkle_exit_power: float = 0.8
|
|
anti_wrinkle_idle_timeout: float = 120.0
|
|
# Dishwasher only: sustained-quiet seconds (after reaching expected duration)
|
|
# that release the end-of-cycle pump-out/drain wait early (#379). Defaults to
|
|
# the shipped constant; per-device configurable so a machine with a long silent
|
|
# passive-drying phase before its final drain can absorb profile drift.
|
|
dishwasher_end_spike_quiet_release: float = DISHWASHER_END_SPIKE_QUIET_RELEASE_SECONDS
|
|
# Fraction of the matched profile's expected (mean) duration Smart Termination
|
|
# requires before it may fire (#393). Device-type-resolved in the manager's
|
|
# config builder (0.99 dishwasher / 0.98 other), so this field always carries a
|
|
# real float - never None - which is what playground.effective_settings() relies
|
|
# on. The dishwasher pump-out relief is combined via min(), so a configured value
|
|
# can only loosen the gate.
|
|
smart_termination_duration_ratio: float = DEFAULT_SMART_TERMINATION_DURATION_RATIO
|
|
# Fraction of the matched profile's expected duration the anti-crease finalise
|
|
# requires before it may fire (#429). A DIFFERENT gate from the ratio above:
|
|
# that one gates Smart Termination, this one the finalise into
|
|
# STATE_ANTI_WRINKLE. Scalar default (no device-type resolution) and always a
|
|
# real float, so playground.effective_settings() never sees None.
|
|
anti_crease_finalize_ratio: float = DEFAULT_ANTI_CREASE_FINALIZE_RATIO
|
|
# How far back readings from aborted start probes may be carried into a
|
|
# committed cycle's curve (#430). 0 disables the whole path, which is the
|
|
# default and keeps the stored-duration convention unchanged.
|
|
curve_preroll_seconds: float = DEFAULT_CURVE_PREROLL_SECONDS
|
|
delay_detect_enabled: bool = False
|
|
# Sustained seconds power must stay in the standby band (between
|
|
# stop_threshold_w and start_threshold_w) before DELAY_WAIT engages.
|
|
# Tuned to filter out brief menu-navigation peaks at the start of a
|
|
# delayed program.
|
|
delay_confirm_seconds: float = 60.0
|
|
delay_timeout_seconds: float = 28800.0
|
|
|
|
|
|
# Add other fields as needed
|
|
|
|
|
|
def trim_zero_readings(
|
|
readings: list[tuple[datetime, float]],
|
|
threshold: float = 0.5,
|
|
trim_start: bool = True,
|
|
trim_end: bool = True,
|
|
) -> list[tuple[datetime, float]]:
|
|
"""Trim continuous zero/near-zero readings from start and end of cycle.
|
|
|
|
Args:
|
|
readings: List of (timestamp, power) tuples
|
|
threshold: Power values below this are considered "zero"
|
|
trim_start: Whether to trim zeros from the beginning
|
|
trim_end: Whether to trim zeros from the end
|
|
|
|
Returns:
|
|
Trimmed list
|
|
"""
|
|
if not readings:
|
|
return readings
|
|
|
|
start_idx = 0
|
|
if trim_start:
|
|
for i, (_, power) in enumerate(readings):
|
|
if power > threshold:
|
|
start_idx = i
|
|
break
|
|
else:
|
|
# All readings are zero - return single point if list not empty
|
|
return readings[:1] if readings else []
|
|
|
|
end_idx = len(readings) - 1
|
|
if trim_end:
|
|
# Find last non-zero reading
|
|
found_end = False
|
|
for i in range(len(readings) - 1, -1, -1):
|
|
if readings[i][1] > threshold:
|
|
end_idx = i
|
|
found_end = True
|
|
break
|
|
|
|
if not found_end and trim_start:
|
|
# If all zeros and trim_start was checked, it would return early.
|
|
# But if safety fallback needed:
|
|
end_idx = start_idx
|
|
elif not found_end and not trim_start:
|
|
# Trimming end but not start, and all zeros?
|
|
# Keep first point
|
|
end_idx = 0
|
|
|
|
# Return trimmed slice (inclusive of end)
|
|
return readings[start_idx : end_idx + 1]
|
|
|
|
|
|
def standby_near_stop_ceiling(stop_threshold_w: float) -> float:
|
|
"""Top of the standby band's near-stop shape (#445 / register item 383).
|
|
|
|
A plateau at or above the stop threshold and at most this high is an
|
|
appliance sitting at its standby draw, not one doing work. Shared by the
|
|
standby-band finalize, the stall display (#452) and its pause catalogue.
|
|
"""
|
|
stop = float(stop_threshold_w)
|
|
return max(STANDBY_BAND_NEAR_STOP_FACTOR * stop, stop + STANDBY_BAND_NEAR_STOP_W)
|
|
|
|
|
|
#: How many of the most recent stored cycles `learned_standby_level_w` reads.
|
|
STANDBY_LEVEL_RECENT_CYCLES = 30
|
|
|
|
|
|
def learned_standby_level_w(
|
|
cycles: Any, stop_threshold_w: float, start_threshold_w: float
|
|
) -> float | None:
|
|
"""Discussion #452: the standby/display level of a two-level appliance, or None.
|
|
|
|
Two measurements the suggestion engine already makes, in order: the level
|
|
every recent cycle was still drawing when it ended (``detect_standby_above_stop``,
|
|
#445), else the resting draw of the clean cycles (``resting_level_w``, item
|
|
455), timed below the near-stop ceiling instead of below stop so that a draw
|
|
just above stop (#452: 4-5 W on a 2.8 W stop) is seen too. None unless the
|
|
level is at least IDLE_MIN_STANDBY_W and below the start threshold, i.e.
|
|
clearly separated from both "off" and a start. Reads the last
|
|
STANDBY_LEVEL_RECENT_CYCLES cycles. Pure, executor-safe, never raises.
|
|
"""
|
|
try:
|
|
# Lazy: suggestion_engine imports Home Assistant helpers this module does not need.
|
|
from .suggestion_engine import ( # noqa: PLC0415
|
|
_cycle_readings,
|
|
detect_standby_above_stop,
|
|
resting_level_w,
|
|
select_clean_cycles,
|
|
)
|
|
|
|
stop = float(stop_threshold_w)
|
|
start = float(start_threshold_w)
|
|
if not (math.isfinite(stop) and math.isfinite(start)) or stop <= 0:
|
|
return None
|
|
recent = [c for c in list(cycles or [])[-STANDBY_LEVEL_RECENT_CYCLES:] if isinstance(c, dict)]
|
|
adv = detect_standby_above_stop(recent, stop)
|
|
if adv is not None:
|
|
level: float | None = float(adv["idle_w"])
|
|
else:
|
|
clean, _excluded = select_clean_cycles(recent, stop_threshold_w=stop)
|
|
points = [p for p in (_cycle_readings(c) for c in clean) if len(p) >= 5]
|
|
level = resting_level_w(points, standby_near_stop_ceiling(stop))
|
|
if level is None or not math.isfinite(level):
|
|
return None
|
|
if level < IDLE_MIN_STANDBY_W or level >= start:
|
|
return None
|
|
return float(level)
|
|
except Exception: # noqa: BLE001 - a display statistic must never break setup
|
|
return None
|
|
|
|
|
|
def terminal_high_for_guards(
|
|
store: Any,
|
|
config: "CycleDetectorConfig",
|
|
cycle_max_power: Any,
|
|
profile_name: str | None,
|
|
) -> tuple[float, ...] | None:
|
|
"""Element 10 of the match tuple: the matched profile's last high-power block.
|
|
|
|
Two consumers with two different bars, and the bar has to travel with the
|
|
block (register item 351):
|
|
|
|
* **anti-crease** (#399) measures against ``anti_wrinkle_max_power``, the
|
|
dryer's "a tumble is below this" level. Only meaningful while anti-wrinkle
|
|
is on, and it returns a TRIPLE.
|
|
* **the standby-band finalise** (#296 / #445) shares the same predicate and
|
|
used to get nothing at all, because element 10 was supplied only when
|
|
anti-wrinkle was enabled and ``DEFAULT_ANTI_WRINKLE_ENABLED`` is False. So
|
|
on a washing machine ``_anticrease_spin_pending`` returned False at once
|
|
and the machine could finalise on the quiet plateau before its final spin,
|
|
recording that spin as a second cycle. The bar here is a share of the
|
|
cycle's own peak, and it is returned as a QUAD so
|
|
``_high_power_seconds_since`` counts live seconds against the same number.
|
|
|
|
**This lives here, module level, because it had two copies.** The manager
|
|
builds the live match tuple and ``playground`` builds the sim's, and
|
|
``end_gate_eval.py`` drives the detector through the Playground - so an arm
|
|
present in only one of them is invisible to every measurement made with that
|
|
harness, which is exactly how item 352 first measured as a no-op. The copies
|
|
had already drifted in their error handling before they were merged.
|
|
|
|
Total by construction: every failure path returns None, which leaves the
|
|
guard exactly as inert as it was. That is the fail-open direction every input
|
|
here takes, and it is also what lets ``playground`` call it directly while
|
|
keeping its own never-raise contract.
|
|
"""
|
|
if not profile_name or store is None:
|
|
return None
|
|
try:
|
|
if config.anti_wrinkle_enabled:
|
|
return store.profile_terminal_high_block(
|
|
profile_name, config.anti_wrinkle_max_power
|
|
)
|
|
if config.device_type not in STANDBY_BAND_FINALIZE_DEVICE_TYPES:
|
|
return None
|
|
ceiling = float(cycle_max_power or 0.0) * STANDBY_BAND_MAX_FRACTION
|
|
if ceiling <= 0:
|
|
return None
|
|
block = store.profile_terminal_high_block(profile_name, ceiling)
|
|
if block is None:
|
|
return None
|
|
return (float(block[0]), float(block[1]), float(block[2]), ceiling)
|
|
except Exception: # noqa: BLE001 - a guard input must never break matching
|
|
return None
|
|
|
|
|
|
def _merge_spans(spans: list[tuple[float, float]]) -> list[tuple[float, float]]:
|
|
"""``(at, length)`` spans sorted and merged where they overlap (item 514: a
|
|
stall shown during a user pause is banked by both)."""
|
|
merged: list[tuple[float, float]] = []
|
|
for at, length in sorted(spans):
|
|
if merged and at <= merged[-1][0] + merged[-1][1]:
|
|
a0, n0 = merged[-1]
|
|
merged[-1] = (a0, max(a0 + n0, at + length) - a0)
|
|
else:
|
|
merged.append((at, length))
|
|
return merged
|
|
|
|
|
|
class CycleDetector:
|
|
"""Detects washing machine cycles based on power usage.
|
|
|
|
Implements a robust state machine:
|
|
OFF -> STARTING -> RUNNING <-> PAUSED -> ENDING -> OFF
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
config: CycleDetectorConfig,
|
|
on_state_change: Callable[[str, str], None],
|
|
on_cycle_end: Callable[[dict[str, Any]], None],
|
|
profile_matcher: (
|
|
Callable[
|
|
[list[tuple[datetime, float]]],
|
|
tuple[str | None, float, float, str | None] | None,
|
|
]
|
|
| None
|
|
) = None,
|
|
device_name: str = "",
|
|
end_confidence_provider: (
|
|
Callable[[list[tuple[float, float]], float], float | None] | None
|
|
) = None,
|
|
terminal_drop_provider: (
|
|
Callable[[list[tuple[float, float]], float], bool | None] | None
|
|
) = None,
|
|
) -> None:
|
|
"""Initialize the cycle detector."""
|
|
self._logger = DeviceLoggerAdapter(_LOGGER, device_name)
|
|
self._config = config
|
|
self._on_state_change = on_state_change
|
|
self._on_cycle_end = on_cycle_end
|
|
self._profile_matcher = profile_matcher
|
|
# Opt-in ML end-guard: (points, expected_duration) -> P(true end) or None.
|
|
# Injected by the manager; None disables the guard (existing behavior).
|
|
self._end_confidence_provider = end_confidence_provider
|
|
# Opt-in terminal-drop detector: (points, expected_duration) -> bool.
|
|
# True means the current low-power event is an anomalously-early hard
|
|
# cliff-to-0 (never seen this early on this device), so the cycle may be
|
|
# finalized without waiting out the full soak-bridging min_off_gap.
|
|
# Injected by the manager; None disables it (existing behavior). Opposite
|
|
# asymmetry to the end-guard: it can only ever *shorten* the end wait.
|
|
self._terminal_drop_provider = terminal_drop_provider
|
|
# Throttle caches for the two providers, scoped to the cycle + expectation:
|
|
# (last_reading_ts, expected_duration, cycle_start, result). Reused only
|
|
# within the recompute window when expected_duration and cycle_start match.
|
|
self._ml_end_cache: tuple[datetime, float, datetime, float | None] | None = None
|
|
self._terminal_drop_cache: tuple[datetime, float, datetime, bool] | None = None
|
|
# Cycle duration (s) at which the ML guard first deferred the current
|
|
# ending episode; bounds how long the guard may keep deferring.
|
|
self._ml_defer_start_duration: float | None = None
|
|
|
|
# State
|
|
self._state = STATE_OFF
|
|
self._sub_state: str | None = None
|
|
self._ignore_power_until_idle: bool = False
|
|
# Sustained high-power time accrued while the stop lockout is armed; used
|
|
# to release the lockout for a genuinely new back-to-back load (#267).
|
|
self._lockout_high_seconds: float = 0.0
|
|
|
|
# Data
|
|
self._power_readings: list[tuple[datetime, float]] = [] # (time, raw_power)
|
|
# Rolling pre-cycle readings, so a start that took several probes to
|
|
# commit can recover the readings the aborted probes took with them
|
|
# (#430). Only appended to while no cycle is open, trimmed to
|
|
# curve_preroll_seconds, and cleared at every cycle end - a previous
|
|
# cycle's tail must never be carried into the next cycle's curve.
|
|
self._preroll_buffer: list[tuple[datetime, float]] = []
|
|
self._current_cycle_start: datetime | None = None
|
|
self._last_active_time: datetime | None = None
|
|
# Last reading the POWER SENSOR actually sent, as opposed to one the
|
|
# manager injected to advance the quiet timers. Deliberately not reset per
|
|
# cycle: it describes the sensor, not the run.
|
|
self._last_real_reading_time: datetime | None = None
|
|
# When the power sensor's state became unavailable / unknown / non-numeric
|
|
# (register item 266): set by the manager, cleared by the next real
|
|
# reading. Sensor state like the field above, so not reset per cycle.
|
|
self._sensor_outage_since: datetime | None = None
|
|
# Traceback of the last restore_state_snapshot that failed, else None.
|
|
self.restore_error: str | None = None
|
|
self._cycle_max_power: float = 0.0
|
|
|
|
# Accumulators (dt-aware)
|
|
self._energy_since_idle_wh: float = 0.0
|
|
self._time_above_threshold: float = 0.0
|
|
self._time_below_threshold: float = 0.0
|
|
# As above, but restarts whenever an outage-sized gap breaks the observed
|
|
# quiet tail, so it counts only quiet WashData actually saw. Used by the
|
|
# dishwasher quiet-release gates so a single low sample after a telemetry
|
|
# dropout can't satisfy them without the quiet having been observed.
|
|
self._time_below_threshold_gapfree: float = 0.0
|
|
# The part of the current below-threshold run that fell inside a sensor
|
|
# outage and so was NOT added to `_time_below_threshold` (item 266). Only
|
|
# the hazard gate reads it, to place the quiet run's start honestly.
|
|
self._time_below_unobserved: float = 0.0
|
|
self._last_process_time: datetime | None = None
|
|
|
|
# New State Machine trackers
|
|
self._state_enter_time: datetime | None = None
|
|
self._matched_profile: str | None = None
|
|
self._verified_pause: bool = False
|
|
self._user_paused: bool = False
|
|
# Item 514: when the current user pause began, and the pauses this cycle
|
|
# already resumed from as (seconds from the cycle start, length).
|
|
self._user_pause_since: datetime | None = None
|
|
self._user_pause_spans: list[tuple[float, float]] = []
|
|
|
|
self._last_power: float | None = None
|
|
self._time_in_state: float = 0.0
|
|
|
|
# Smoothing buffer
|
|
|
|
# Adaptive Sampling Tracker
|
|
self._recent_dts: list[float] = [] # Track last 20 dt values
|
|
self._p95_dt: float = 1.0 # Default assumption
|
|
# The cadence as it stood BEFORE the reading currently being processed was
|
|
# folded in. Every gap-vs-outage classification reads this, never _p95_dt:
|
|
# an outage-sized interval that has already widened p95 would raise the very
|
|
# ceiling meant to catch it (a 120 s gap after a 10 s cadence lifts p95 to
|
|
# ~15.5 s -> ceiling 155 s -> the gap passes as observed time).
|
|
self._prior_p95_dt: float = 1.0
|
|
|
|
# Profile Matching Tracker
|
|
self._last_match_time: datetime | None = None
|
|
# Whether the manager has committed a program this cycle (set_match_committed).
|
|
self._match_committed: bool = False
|
|
self._expected_duration: float = 0.0
|
|
self._last_match_confidence: float = 0.0
|
|
# Element 12: the longest expected duration among the candidates the
|
|
# matcher still considers plausible. An ambiguous match may mean one of
|
|
# them is materially longer than the winner, so "past the expected end"
|
|
# might be "mid-soak in that longer programme" - but only up to THIS
|
|
# duration. Past it there is no longer programme left to be mid-soak in,
|
|
# and the guard's own rationale is spent.
|
|
self._longest_candidate_duration: float = 0.0
|
|
self._end_spike_seen: bool = False
|
|
self._end_spike_duration: float = 0.0 # cycle duration (s) when _end_spike_seen was last set
|
|
self._match_ambiguous: bool = False # last live match was ambiguous (gates predictive end)
|
|
# The #288 prefix-landscape term (element 8): a much longer candidate with
|
|
# a good full-envelope shape exists. Read only by the anti-crease finalize
|
|
# (see _anticrease_gate_open). The ENDING gates' own prefix flag (#364
|
|
# prefix fit, element 7) was removed in 0.5.8.
|
|
self._match_prefix_ambiguous_full_shape: bool = False
|
|
# Mean power the matched profile draws over the last few % of its own run
|
|
# (profile_store.profile_tail_power). None = no opinion, guard stays inert.
|
|
self._matched_tail_power: float | None = None
|
|
# (start_frac, seconds, start_offset_s) of the matched profile's own terminal
|
|
# high-power block, from profile_store.profile_terminal_high_block. None = the
|
|
# profile has no such block (or we have no opinion) and the guard stays inert
|
|
# (#399). A two-element payload (pre-item-196 snapshot, Playground, older
|
|
# callers) is still accepted and falls back to start_frac x expected.
|
|
self._matched_terminal_high: (
|
|
tuple[float, float] | tuple[float, float, float] | None
|
|
) = None
|
|
# Element 11 (register item 297): how long the matched profile is MEASURED to stay quiet
|
|
# after its last real activity. Bounds how much of Smart Termination's
|
|
# confirmation delay may be banked into the stored duration. None means the
|
|
# profile has not been measured, and for a DISHWASHER the previous
|
|
# expected-end cap applies. Only a dishwasher gets that far: every other
|
|
# device type returns `_last_active_time` from `_keep_tail_cap` before
|
|
# this element is read at all, because nothing legitimate follows their
|
|
# last activity - so for them the cap can sit EARLIER than expected_end.
|
|
self._matched_terminal_quiet_s: float | None = None
|
|
# Element 13 (register item 384): the shortest length the user has vouched
|
|
# for in the matched profile (a corrected `manual_duration`, a recorder
|
|
# capture, a golden cycle). A dishwasher's kept tail never ends before
|
|
# TRUSTED_LENGTH_FLOOR_FRAC of it - the same floor the banked-tail repair
|
|
# applies - because element 11 can be wrong in the direction that deletes
|
|
# a drying phase. None: nothing vouched, no floor.
|
|
self._matched_trusted_min_s: float | None = None
|
|
# Element 14 (audit DETECT-16): the matched profile's pause catalogue,
|
|
# (traced cycles, ((start_fraction, seconds), ...)), for the hazard end
|
|
# gate. None: no catalogue, the fallback waits as before.
|
|
self._matched_pause_catalogue: tuple[int, tuple[tuple[float, float], ...]] | None = None
|
|
# Register item 498: the longest pause in the catalogue of every match
|
|
# revoked this cycle, which bounds a verified pause the revoke orphaned
|
|
# (`_release_orphaned_pause`). 0.0: no evidence, the floor applies.
|
|
self._orphaned_pause_evidence_s: float = 0.0
|
|
self._terminal_quiet_memo: tuple[Any, bool] | None = None
|
|
# One-shot per cycle, so the held-finalise reason is visible in the log
|
|
# without repeating it on every reading.
|
|
self._anticrease_spin_wait_logged: bool = False
|
|
# Register item 393a: set while STATE_ANTI_WRINKLE was entered by the #296
|
|
# anti-crease finalise - the level at or under which a reading is that
|
|
# tail's own baseline rather than a burst. None for every other entry.
|
|
self._anticrease_tail_floor_w: float | None = None
|
|
self._last_smart_term_block_reason: str | None = None # #346 diagnostic throttle
|
|
|
|
# Anti-wrinkle tracking (dryers only)
|
|
self._anti_wrinkle_candidate_start: datetime | None = None
|
|
self._anti_wrinkle_candidate_peak: float = 0.0
|
|
self._anti_wrinkle_candidate_start_power: float = 0.0
|
|
self._anti_wrinkle_idle_time: float = 0.0 # Track time spent below exit_power while in ANTI_WRINKLE
|
|
|
|
# Delayed-start band tracking.
|
|
# _delay_band_start anchors the first reading in the standby band
|
|
# [stop_threshold_w, start_threshold_w) while still in STATE_OFF.
|
|
# _delay_band_seconds mirrors the anchored elapsed time for
|
|
# diagnostics and tests.
|
|
self._delay_band_start: datetime | None = None
|
|
self._delay_band_seconds: float = 0.0
|
|
# _delay_band_peak is purely diagnostic - surfaced in the log line
|
|
# when the transition fires so users can see what their machine's
|
|
# actual standby plateau looked like.
|
|
self._delay_band_peak: float = 0.0
|
|
# _delay_wait_true_off_seconds tracks sustained "true off" (power
|
|
# below stop_threshold_w) inside DELAY_WAIT, so we can drop back to
|
|
# OFF only when the machine has clearly been switched off rather
|
|
# than briefly dipped.
|
|
self._delay_wait_true_off_seconds: float = 0.0
|
|
# _starting_paused_off_since anchors the first below-stop reading while
|
|
# a user-paused STARTING state is held (issue #306). Elapsed time is
|
|
# measured from this anchor, not accumulated per-dt, so a single large
|
|
# (but sub-outage) interval cannot prematurely credit minutes of quiet time.
|
|
self._starting_paused_off_since: datetime | None = None
|
|
# _delay_wait_high_start anchors the first high-power reading
|
|
# observed inside DELAY_WAIT. We only transition to STARTING
|
|
# when the high-power streak has lasted at least
|
|
# start_duration_threshold real seconds - measured between two
|
|
# consecutive high readings, not from the dt to the previous
|
|
# (low) reading. This prevents a single isolated spike from
|
|
# tripping STARTING just because the sampling interval is long.
|
|
self._delay_wait_high_start: datetime | None = None
|
|
self._delay_wait_high_power: float | None = None
|
|
# Preserve a delayed-start candidate across a false STARTING probe
|
|
# that drops back into the standby band without the machine truly
|
|
# turning off.
|
|
self._preserve_delay_band_on_off: bool = False
|
|
# Register item 504: when the current STARTING probe came out of DELAY_WAIT,
|
|
# the moment DELAY_WAIT was entered (its timeout anchor). A false start that
|
|
# falls back into the standby band returns to DELAY_WAIT with this anchor
|
|
# instead of to OFF, so a standby that straddles start_threshold_w keeps
|
|
# waiting until a real start, a true off, or the band's own timeout.
|
|
# None for every other probe. In memory only: a probe restored after a
|
|
# restart falls back to OFF as before.
|
|
self._probe_wait_since: datetime | None = None
|
|
# Item 515: when the current probe came out of a terminal state, that state,
|
|
# its entry time and its sub-state, for the false start to return to. None
|
|
# for every other probe; in memory only (a restored probe falls to OFF).
|
|
self._probe_terminal_from: tuple[str, datetime | None, str | None] | None = None
|
|
# Register item 501: a standby that straddles start_threshold_w (#35: 2-22 W
|
|
# around a 4.24 W threshold) re-probes on almost every reading.
|
|
# `_standby_reprobe` is set when a false start falls back into the band
|
|
# (>= stop_threshold_w) or a terminal state reads the band (item 510), and
|
|
# cleared by any reading below it and by entering any state other than
|
|
# OFF/STARTING. A probe that begins while it is set, or out
|
|
# of DELAY_WAIT (a standby by definition), is HIDDEN: it runs exactly as any
|
|
# other, but `exposed_state` keeps showing the state it began in until it
|
|
# has filled STANDBY_REPROBE_SHOW_ENERGY_FRACTION of the energy gate.
|
|
self._standby_reprobe: bool = False
|
|
self._probe_hidden: bool = False
|
|
self._probe_hidden_state: str = STATE_OFF
|
|
self._probe_hidden_sub_state: str | None = None
|
|
# Discussion #452, display only (see the STALL_* / IDLE_* constants).
|
|
# Element 15: the matched profile's low stretches below the near-stop
|
|
# ceiling, as element 14's (traced, ((fraction, seconds), ...)).
|
|
self._matched_stall_catalogue: tuple[int, tuple[tuple[float, float], ...]] | None = None
|
|
# The current flat near-stop run: its first reading and its spread.
|
|
self._stall_run_start: datetime | None = None
|
|
self._stall_run_lo: float = 0.0
|
|
self._stall_run_hi: float = 0.0
|
|
self._stall_active: bool = False
|
|
# The match as it stood when the current run began (`_begin_stall_run`):
|
|
# a plateau soon looks like a finished SHORTER programme to the matcher,
|
|
# so the evidence is the pre-plateau match's. And (required seconds,
|
|
# owes work, owes its terminal high-power block), computed from it once.
|
|
self._stall_match: tuple[Any, ...] | None = None
|
|
self._stall_eval: tuple[float, bool, bool] | None = None
|
|
# ...and whether the CURRENT match agrees, re-read after every match.
|
|
self._stall_now: tuple[bool, bool] | None = None
|
|
# Item 514: the current run began straight out of activity.
|
|
self._stall_run_abrupt: bool = False
|
|
# Item 511: the runs this cycle ended while shown as stalled, as (elapsed
|
|
# seconds at the run's first reading, run seconds), in order.
|
|
self._stall_spans: list[tuple[float, float]] = []
|
|
# The learned standby level (`learned_standby_level_w`, set by the
|
|
# manager) and the debounced idle/off class it drives.
|
|
self._standby_level_w: float | None = None
|
|
self._idle_shown: bool = False
|
|
self._idle_candidate_since: datetime | None = None
|
|
# A LEARNED level shows idle only once the appliance has been seen at its
|
|
# off level outside a cycle (held IDLE_DEBOUNCE_S) since start-up: an
|
|
# appliance whose switched-off draw reaches the level would otherwise read
|
|
# idle for ever, and an automation waiting for `off` would never fire.
|
|
self._idle_off_seen: bool = False
|
|
self._idle_off_since: datetime | None = None
|
|
|
|
@property
|
|
def exposed_state(self) -> str:
|
|
"""The state entities show. Detection never reads this.
|
|
|
|
``state``, except: during a hidden probe the state it began in (item 501:
|
|
a standby that keeps crossing the start threshold otherwise wrote ~500
|
|
off/starting rows a day and fired every automation keyed on
|
|
``starting``); ``paused`` while the open cycle is stalled (#452); and
|
|
``idle`` instead of ``off`` while a two-level appliance sits at its
|
|
standby level (#452).
|
|
"""
|
|
state = self._state
|
|
if self._probe_hidden and state == STATE_STARTING:
|
|
state = self._probe_hidden_state
|
|
if state == STATE_OFF and self._idle_shown and self._idle_bounds() is not None:
|
|
return STATE_IDLE
|
|
if self._stall_active and state in (STATE_RUNNING, STATE_ENDING):
|
|
return STATE_PAUSED
|
|
return state
|
|
|
|
@property
|
|
def exposed_sub_state(self) -> str | None:
|
|
"""``sub_state`` to show, matching :attr:`exposed_state`."""
|
|
exposed = self.exposed_state
|
|
if exposed == STATE_IDLE:
|
|
return STATE_IDLE.capitalize()
|
|
if self._stall_active and exposed == STATE_PAUSED:
|
|
return STALL_SUB_STATE
|
|
if self._probe_hidden and self._state == STATE_STARTING:
|
|
return self._probe_hidden_sub_state
|
|
return self._sub_state
|
|
|
|
@property
|
|
def stalled(self) -> bool:
|
|
"""Whether the open cycle is stalled (#452). Display only."""
|
|
return self._stall_active
|
|
|
|
def stall_info(self) -> dict[str, Any] | None:
|
|
"""Small, event-safe description of the current stall, or None."""
|
|
if not self._stall_active or self._stall_run_start is None:
|
|
return None
|
|
start = self._current_cycle_start
|
|
last = self._last_process_time or self._stall_run_start
|
|
return {
|
|
"stalled_since": self._stall_run_start.isoformat(),
|
|
"stalled_for_s": round(max(0.0, (last - self._stall_run_start).total_seconds())),
|
|
"elapsed_at_stall_s": (
|
|
round((self._stall_run_start - start).total_seconds()) if start else None
|
|
),
|
|
"plateau_w": round((self._stall_run_lo + self._stall_run_hi) / 2.0, 2),
|
|
# The match the halt began under when the halt has since cost it.
|
|
"matched_profile": self._matched_profile or (
|
|
self._stall_match[0] if self._stall_match else None
|
|
),
|
|
}
|
|
|
|
def set_standby_level(self, level_w: float | None) -> None:
|
|
"""The learned standby level for the idle display (#452); None: none."""
|
|
try:
|
|
level = float(level_w) if level_w is not None else None
|
|
except (TypeError, ValueError, OverflowError):
|
|
level = None
|
|
self._standby_level_w = level if level is not None and math.isfinite(level) else None
|
|
|
|
def _idle_bounds(self) -> tuple[float, float, float, bool] | None:
|
|
"""``(enter_w, exit_w, debounce_s, off level confirmed)`` of the idle
|
|
display, or None (#452)."""
|
|
cfg = self._config
|
|
start = float(cfg.start_threshold_w)
|
|
pot = cfg.power_off_threshold_w
|
|
if isinstance(pot, (int, float)) and 0.0 < float(pot) < float(cfg.stop_threshold_w):
|
|
# The user's own off level (#284): the same boundary and debounce the
|
|
# power-based Off uses, so a reset to off never shows idle first.
|
|
return (float(pot), float(pot), max(0.0, float(cfg.power_off_delay)), True)
|
|
level = self._standby_level_w
|
|
if level is None or level < IDLE_MIN_STANDBY_W or level >= start:
|
|
return None
|
|
enter = IDLE_OFF_FRACTION * level
|
|
return (enter, IDLE_EXIT_FRACTION * enter, IDLE_DEBOUNCE_S, False)
|
|
|
|
def _update_idle(self, timestamp: datetime, power: float) -> None:
|
|
"""Debounced idle/off class of the latest real reading (#452). O(1).
|
|
|
|
A reading at or above the enter level votes idle, one below the exit
|
|
level votes off, one in between keeps the current class. The class
|
|
changes only once the other vote has held for the debounce, timed from
|
|
its first reading, so an excursion shorter than that never shows.
|
|
"""
|
|
bounds = self._idle_bounds()
|
|
if bounds is None:
|
|
self._idle_shown = False
|
|
self._idle_candidate_since = None
|
|
return
|
|
enter, exit_w, debounce, confirmed = bounds
|
|
if power >= enter:
|
|
vote = True
|
|
elif power < exit_w:
|
|
vote = False
|
|
else:
|
|
vote = self._idle_shown
|
|
if power < exit_w and self._state in (
|
|
STATE_OFF, STATE_FINISHED, STATE_INTERRUPTED, STATE_FORCE_STOPPED
|
|
):
|
|
if self._idle_off_since is None:
|
|
self._idle_off_since = timestamp
|
|
if (timestamp - self._idle_off_since).total_seconds() >= debounce:
|
|
self._idle_off_seen = True
|
|
else:
|
|
self._idle_off_since = None
|
|
if vote and not (confirmed or self._idle_off_seen):
|
|
vote = False # never seen switched off: not yet known to be two-level
|
|
if vote == self._idle_shown:
|
|
self._idle_candidate_since = None
|
|
return
|
|
if self._idle_candidate_since is None:
|
|
self._idle_candidate_since = timestamp
|
|
if (timestamp - self._idle_candidate_since).total_seconds() >= debounce:
|
|
self._idle_shown = vote
|
|
self._idle_candidate_since = None
|
|
|
|
@classmethod
|
|
def _sanitize_stall_catalogue(cls, raw: Any) -> Any:
|
|
"""Element 15, keeping only the stretches that can set a stall's wait.
|
|
|
|
The near-stop catalogue holds every gap between two drum tumbles; a
|
|
stretch under STALL_MIN_S / END_GATE_HAZARD_MARGIN can never raise the
|
|
wait above STALL_MIN_S, so it is dropped. A callable (the producers' lazy
|
|
form) is kept as is and resolved by :meth:`_resolve_stall_catalogue`.
|
|
"""
|
|
if callable(raw):
|
|
return raw
|
|
cat = cls._sanitize_pause_catalogue(raw)
|
|
if cat is None:
|
|
return None
|
|
floor = STALL_MIN_S / END_GATE_HAZARD_MARGIN
|
|
return (cat[0], tuple(p for p in cat[1] if p[1] >= floor))
|
|
|
|
@staticmethod
|
|
def _sanitize_stall_spans(raw: Any) -> list[tuple[float, float]]:
|
|
"""Item 511's finished stalls from a snapshot: finite, non-negative pairs
|
|
in order; anything else is dropped (an old snapshot has none)."""
|
|
return match_rules.sanitize_stall_spans(raw)
|
|
|
|
@classmethod
|
|
def _resolve_stall_catalogue(
|
|
cls, raw: Any
|
|
) -> tuple[int, tuple[tuple[float, float], ...]] | None:
|
|
"""Element 15 as data: a lazy producer is called here, once per run."""
|
|
if callable(raw):
|
|
try:
|
|
raw = raw()
|
|
except Exception: # noqa: BLE001 - missing evidence, not an error
|
|
return None
|
|
return cls._sanitize_stall_catalogue(raw) if raw is not None else None
|
|
|
|
def _seed_stall_run(self) -> None:
|
|
"""Re-derive the flat near-stop run in progress from the trace (#452)."""
|
|
self._clear_stall()
|
|
if self._state not in (STATE_RUNNING, STATE_PAUSED, STATE_ENDING):
|
|
return
|
|
band = self._stall_band()
|
|
if band is None:
|
|
return
|
|
lo = hi = None
|
|
start = None
|
|
for ts, p in reversed(self._power_readings):
|
|
p = float(p)
|
|
if not band[0] <= p <= band[1]:
|
|
break
|
|
nlo = p if lo is None else min(lo, p)
|
|
nhi = p if hi is None else max(hi, p)
|
|
if nhi - nlo > band[2]:
|
|
break
|
|
lo, hi, start = nlo, nhi, ts
|
|
if start is not None and lo is not None and hi is not None:
|
|
# The restored match stands in for the one the run began under.
|
|
self._begin_stall_run(start, lo)
|
|
self._stall_run_hi = hi
|
|
|
|
def _gate_spans(self) -> list[tuple[float, float]]:
|
|
"""The spans the gates' programme time leaves out, in order and merged:
|
|
every finished stall (item 511) and every finished user pause (item 514)."""
|
|
stalls = self._stall_spans if STALL_EXCLUDED_FROM_GATES else []
|
|
pauses = self._user_pause_spans if USER_PAUSE_EXCLUDED_FROM_GATES else []
|
|
if not pauses:
|
|
return stalls
|
|
if not stalls:
|
|
return pauses
|
|
return _merge_spans([*stalls, *pauses])
|
|
|
|
def _gate_elapsed_s(self, timestamp: datetime) -> float:
|
|
"""Elapsed seconds the duration-based end gates read (item 511): the
|
|
cycle's wall clock less every finished stall and user pause (item 514).
|
|
0.0 with no cycle open."""
|
|
start = self._current_cycle_start
|
|
if start is None:
|
|
return 0.0
|
|
elapsed = (timestamp - start).total_seconds()
|
|
spans = self._gate_spans()
|
|
if spans:
|
|
elapsed = max(0.0, elapsed - sum(length for _at, length in spans))
|
|
return elapsed
|
|
|
|
def _wall_offset_s(self, offset_s: float) -> float:
|
|
"""A programme-time offset from the cycle start as wall-clock seconds
|
|
(item 511): each left-out span that began before it moves it later."""
|
|
shift = 0.0
|
|
for at, length in self._gate_spans():
|
|
if offset_s < at - shift:
|
|
break
|
|
shift += length
|
|
return offset_s + shift
|
|
|
|
def _match_readings(self) -> list[tuple[datetime, float]]:
|
|
"""The trace the live matcher reads (item 511): each finished stall cut
|
|
out and every later reading moved back by it, so the duration, the
|
|
Stage-1 ratio and the in-progress Stage-4 stretch are programme time and
|
|
the shape has no plateau the programme never made. A stall still shown
|
|
stays in, as before: its evidence includes the current match's reading
|
|
of it. The stored trace keeps every reading."""
|
|
return self._cut_readings(self._gate_spans())
|
|
|
|
def _cut_readings(
|
|
self, spans: list[tuple[float, float]]
|
|
) -> list[tuple[datetime, float]]:
|
|
"""The trace with ``spans`` cut out, every later reading moved back by them
|
|
(`match_rules.stall_cut_plan`). The trace itself when there is none."""
|
|
start = self._current_cycle_start
|
|
if not spans or start is None:
|
|
return self._power_readings
|
|
readings = self._power_readings
|
|
plan = match_rules.stall_cut_plan(
|
|
[(ts - start).total_seconds() for ts, _p in readings], spans
|
|
)
|
|
return [
|
|
(readings[i][0] - timedelta(seconds=shift), readings[i][1]) if shift
|
|
else readings[i]
|
|
for i, shift in plan
|
|
]
|
|
|
|
def _progress_spans(
|
|
self, timestamp: datetime | None
|
|
) -> tuple[list[tuple[float, float]], tuple[float, float] | None]:
|
|
"""``(finished stalls, the stall shown now as (at, seconds so far) or None)``
|
|
that progress leaves out (item 514)."""
|
|
if not STALL_EXCLUDED_FROM_PROGRESS:
|
|
return [], None
|
|
start, run = self._current_cycle_start, self._stall_run_start
|
|
current = None
|
|
if (
|
|
STALL_CURRENT_EXCLUDED_FROM_PROGRESS and self._stall_active
|
|
and start is not None and run is not None and timestamp is not None
|
|
):
|
|
current = (
|
|
max(0.0, (run - start).total_seconds()),
|
|
max(0.0, (dt_util.as_utc(timestamp) - run).total_seconds()),
|
|
)
|
|
return list(self._stall_spans), current
|
|
|
|
def progress_elapsed_s(self, elapsed_s: float, timestamp: datetime | None) -> float:
|
|
"""``elapsed_s`` less the stalled time progress leaves out (item 514).
|
|
|
|
``elapsed_s`` is the caller's own clock: the manager's net of the user
|
|
pause, the Playground's replay offset. ``timestamp`` is now (the stall
|
|
shown at this moment counts up to it)."""
|
|
finished, current = self._progress_spans(timestamp)
|
|
stalled = sum(length for _at, length in finished)
|
|
if current is not None:
|
|
stalled += current[1]
|
|
return max(0.0, float(elapsed_s) - stalled)
|
|
|
|
def progress_trace(self, timestamp: datetime | None) -> list[tuple[datetime, float]]:
|
|
"""The trace progress reads (item 514): the live matcher's, and with a
|
|
stall shown now everything from its first reading on is left out too.
|
|
A copy, like :meth:`get_power_trace`."""
|
|
finished, current = self._progress_spans(timestamp)
|
|
if current is not None:
|
|
finished = [*finished, (current[0], math.inf)]
|
|
return list(self._cut_readings(finished))
|
|
|
|
def _bank_stall(self, timestamp: datetime) -> None:
|
|
"""Keep the run that ends at ``timestamp`` if it is shown as stalled."""
|
|
start, run_start = self._current_cycle_start, self._stall_run_start
|
|
if self._stall_active and start is not None and run_start is not None:
|
|
self._stall_spans.append((
|
|
max(0.0, (run_start - start).total_seconds()),
|
|
max(0.0, (timestamp - run_start).total_seconds()),
|
|
))
|
|
|
|
def _clear_stall(self) -> None:
|
|
self._stall_run_start = None
|
|
self._stall_run_lo = self._stall_run_hi = 0.0
|
|
self._stall_active = False
|
|
self._stall_match = None
|
|
self._stall_eval = None
|
|
self._stall_now = None
|
|
self._stall_run_abrupt = False
|
|
|
|
def _begin_stall_run(self, timestamp: datetime, power: float) -> None:
|
|
"""A new flat near-stop run starts at this reading: snapshot the match."""
|
|
self._stall_run_start = timestamp
|
|
self._stall_run_lo = self._stall_run_hi = power
|
|
self._stall_eval = None
|
|
self._stall_now = None
|
|
# Item 514: straight out of activity, judged on the reading before this
|
|
# one (a restart re-derives it from the restored trace).
|
|
self._stall_run_abrupt = False
|
|
band = self._stall_band() if STALL_ABRUPT_PEAK_FRACTION > 0 else None
|
|
if band is not None:
|
|
prev = next(
|
|
(float(p) for ts, p in reversed(self._power_readings) if ts < timestamp), None
|
|
)
|
|
self._stall_run_abrupt = prev is not None and prev >= max(
|
|
STALL_ABRUPT_PEAK_FRACTION * float(self._cycle_max_power), 2.0 * band[1]
|
|
)
|
|
self._stall_match = (
|
|
self._matched_profile if self._expected_duration > 0 else None,
|
|
float(self._expected_duration),
|
|
self._matched_terminal_high,
|
|
self._matched_stall_catalogue,
|
|
)
|
|
|
|
def _stall_evidence(self) -> tuple[float, bool, bool] | None:
|
|
"""``(required s, owes work, owes its spin)`` of the current run, or None."""
|
|
if self._stall_run_start is None:
|
|
return None
|
|
if self._stall_eval is None:
|
|
self._stall_eval = self._stall_evaluate()
|
|
return self._stall_eval
|
|
|
|
def _stall_holds_standby_band(self) -> bool:
|
|
"""Whether the current flat run says the programme is not done (#452).
|
|
|
|
Only the strongest evidence holds a finalize: the match the run began
|
|
under ends on a terminal high-power block (#399) this cycle has not
|
|
produced yet. A matcher that re-reads the halt as a finished SHORTER
|
|
programme cannot release it, and the plain position test the display also
|
|
accepts holds nothing (a programme matched to a longer one ends "early"
|
|
every time). Independent of the display's wait: the finalize's own 10 min
|
|
window and expected-duration gate already say "long enough".
|
|
"""
|
|
if not STALL_HOLDS_STANDBY_BAND or self._stall_band() is None:
|
|
return False
|
|
evidence = self._stall_evidence()
|
|
if self._stall_abrupt_owes():
|
|
return True # item 514: an abrupt halt under a match that owes work
|
|
return bool(evidence and evidence[2] and self._stall_now_evidence()[1])
|
|
|
|
def _stall_abrupt_owes(self) -> bool:
|
|
"""Item 514: the run began straight out of activity under a MATCHED
|
|
programme that still owes work, so the current match's verdict (the
|
|
plateau read as a finished shorter programme) is not needed."""
|
|
if not (self._stall_run_abrupt and self._stall_match and self._stall_match[0]):
|
|
return False
|
|
evidence = self._stall_evidence()
|
|
return bool(evidence and evidence[1])
|
|
|
|
def _stall_now_evidence(self) -> tuple[bool, bool]:
|
|
"""``(owes work, owes its spin)`` by the CURRENT match, for the current run.
|
|
|
|
Both matches must agree before anything shows or holds: the one the run
|
|
began under can be a prefix match of a longer programme (the last live
|
|
tick before a real end differs from the complete match on 17.5% of
|
|
cycles, audit MATCH-DECIDE-02), and the current one can be the halt read
|
|
as a finished shorter programme. Measured with ``end_gate_eval.py
|
|
--halt-at``: requiring both cut real ends held under a 45 min display-on
|
|
tail 15 -> 6 of 258 and kept 28 of the 48 halts the pre-plateau match
|
|
alone kept open. With no current match the pre-plateau verdict stands.
|
|
"""
|
|
if self._stall_now is not None:
|
|
return self._stall_now
|
|
start, run_start = self._current_cycle_start, self._stall_run_start
|
|
expected = float(self._expected_duration)
|
|
block = self._matched_terminal_high
|
|
if not (self._matched_profile and expected > 0 and start and run_start):
|
|
evidence = self._stall_evidence()
|
|
now = (bool(evidence and evidence[1]), bool(evidence and evidence[2]))
|
|
elif block is not None and block[0] >= ANTI_CREASE_TERMINAL_HIGH_MIN_FRAC:
|
|
needed = float(block[1]) * ANTI_CREASE_TERMINAL_MATCH_FRAC
|
|
offset_s = float(block[2]) if len(block) >= 3 else float(block[0]) * expected
|
|
ceiling_w = float(block[3]) if len(block) >= 4 else None
|
|
spin = needed > 0 and self._high_power_seconds_since(
|
|
offset_s, ceiling_w=ceiling_w
|
|
) < needed
|
|
now = (spin, spin)
|
|
else:
|
|
# Programme time (item 511): an earlier stall is not progress.
|
|
position = self._gate_elapsed_s(run_start) / expected
|
|
now = (position < STALL_OWES_MAX_POSITION, False)
|
|
self._stall_now = now
|
|
return now
|
|
|
|
def _set_stalled(self, stalled: bool, timestamp: datetime) -> None:
|
|
"""Flip the stall flag (one place, so the eval harness can observe it)."""
|
|
if stalled == self._stall_active:
|
|
return
|
|
self._stall_active = stalled
|
|
if stalled:
|
|
self._logger.info(
|
|
"Cycle stalled: flat %.1f-%.1f W (stop %.2f W) for %.0fs, matched %s; "
|
|
"shown as paused (display only, the cycle stays open)",
|
|
self._stall_run_lo,
|
|
self._stall_run_hi,
|
|
self._config.stop_threshold_w,
|
|
(timestamp - (self._stall_run_start or timestamp)).total_seconds(),
|
|
self._matched_profile,
|
|
)
|
|
else:
|
|
self._logger.debug("Cycle no longer stalled at %s", timestamp)
|
|
|
|
def _stall_band(self) -> tuple[float, float, float] | None:
|
|
"""``(low_w, high_w, flatness_w)`` a stall plateau must sit in, or None."""
|
|
if self._config.device_type not in STALL_DEVICE_TYPES:
|
|
return None
|
|
stop = float(self._config.stop_threshold_w)
|
|
peak = float(self._cycle_max_power)
|
|
if stop <= 0 or peak <= 0:
|
|
return None
|
|
high = min(standby_near_stop_ceiling(stop), peak * STANDBY_BAND_MAX_FRACTION)
|
|
if high < stop:
|
|
return None
|
|
flat = max(STANDBY_BAND_FLATNESS_FLOOR_W, peak * STANDBY_BAND_FLATNESS_FRACTION)
|
|
return (stop, high, flat)
|
|
|
|
def _stall_evaluate(self) -> tuple[float, bool, bool]:
|
|
"""``(required seconds, owes work, owes its spin)`` for the current run,
|
|
from the match it began under."""
|
|
start = self._current_cycle_start
|
|
run_start = self._stall_run_start
|
|
name, expected, block, cat = self._stall_match or (None, 0.0, None, None)
|
|
if not (name and expected > 0 and start and run_start):
|
|
return (STALL_UNMATCHED_MIN_S, True, False)
|
|
cat = self._resolve_stall_catalogue(cat)
|
|
position = max(0.0, self._gate_elapsed_s(run_start) / expected) # item 511
|
|
# Owes work: the terminal high-power block (#399) not yet produced, or,
|
|
# for a programme without one, a run that began well before its end.
|
|
owes_spin = False
|
|
if block is not None and block[0] >= ANTI_CREASE_TERMINAL_HIGH_MIN_FRAC:
|
|
needed = float(block[1]) * ANTI_CREASE_TERMINAL_MATCH_FRAC
|
|
offset_s = float(block[2]) if len(block) >= 3 else float(block[0]) * expected
|
|
ceiling_w = float(block[3]) if len(block) >= 4 else None
|
|
owes_spin = needed > 0 and self._high_power_seconds_since(
|
|
offset_s, ceiling_w=ceiling_w
|
|
) < needed
|
|
owes = owes_spin
|
|
else:
|
|
owes = position < STALL_OWES_MAX_POSITION
|
|
# Ambiguity is not a reason to refuse here: the display only waits, it
|
|
# never shortens anything (unlike the hazard gate this mirrors).
|
|
if cat is None or cat[0] < END_GATE_HAZARD_MIN_CYCLES:
|
|
return (STALL_UNMATCHED_MIN_S, owes, owes_spin)
|
|
later = [d for f, d in cat[1] if f >= position - END_GATE_HAZARD_POSITION_SLACK]
|
|
need = max(STALL_MIN_S, END_GATE_HAZARD_MARGIN * max(later, default=0.0))
|
|
return (need, owes, owes_spin)
|
|
|
|
def _update_stall(self, timestamp: datetime, power: float) -> None:
|
|
"""Track the flat near-stop run of an open cycle and flag a stall (#452).
|
|
|
|
O(1) per reading; the evidence is computed once per run and match.
|
|
"""
|
|
band = self._stall_band()
|
|
if band is None or not (band[0] <= power <= band[1]):
|
|
if self._stall_run_start is not None or self._stall_active:
|
|
self._bank_stall(timestamp) # item 511
|
|
self._set_stalled(False, timestamp)
|
|
self._clear_stall()
|
|
return
|
|
if self._stall_run_start is None:
|
|
self._begin_stall_run(timestamp, power)
|
|
else:
|
|
lo = min(self._stall_run_lo, power)
|
|
hi = max(self._stall_run_hi, power)
|
|
if hi - lo > band[2] and not self._stall_active:
|
|
# Not flat: a new run starts here. A stall already shown stays
|
|
# shown while readings stay in the band.
|
|
self._begin_stall_run(timestamp, power)
|
|
else:
|
|
self._stall_run_lo, self._stall_run_hi = lo, hi
|
|
run_s = (timestamp - self._stall_run_start).total_seconds()
|
|
if self._stall_eval is None and run_s < min(STALL_MIN_S, STALL_UNMATCHED_MIN_S):
|
|
return
|
|
evidence = self._stall_evidence()
|
|
if evidence is not None:
|
|
self._set_stalled(
|
|
run_s >= evidence[0] and evidence[1]
|
|
and (self._stall_now_evidence()[0] or self._stall_abrupt_owes()),
|
|
timestamp,
|
|
)
|
|
|
|
def _hide_probe(self, hidden: bool) -> None:
|
|
"""Item 501: whether the probe about to begin is hidden. Called BEFORE the
|
|
transition to STARTING, whose callback already refreshes the entities."""
|
|
self._probe_hidden = hidden
|
|
self._probe_hidden_state = self._state
|
|
self._probe_hidden_sub_state = self._sub_state
|
|
|
|
def _return_probe_to_terminal(self, timestamp: datetime, power: float) -> None:
|
|
"""Item 515: a false start goes back to the terminal state it began in.
|
|
|
|
With that state's entry time and sub-state, and without the probe's
|
|
readings, as ``reset()`` leaves a terminal state: no cycle is open, so a
|
|
later force-stop or "Done now" there records nothing. A reading still in
|
|
the band marks the next probe as a standby re-probe (items 501, 510).
|
|
"""
|
|
back = self._probe_terminal_from
|
|
if back is None:
|
|
return
|
|
state, entered, sub_state = back
|
|
self._transition_to(state, timestamp)
|
|
if entered is not None:
|
|
self._state_enter_time = entered
|
|
self._time_in_state = max(0.0, (timestamp - entered).total_seconds())
|
|
self._sub_state = sub_state
|
|
self._power_readings = []
|
|
self._current_cycle_start = None
|
|
self._last_active_time = None
|
|
self._cycle_max_power = 0.0
|
|
self._energy_since_idle_wh = 0.0
|
|
self._time_above_threshold = 0.0
|
|
self._standby_reprobe = power >= self._config.stop_threshold_w
|
|
|
|
def _show_probe_with_evidence(self) -> None:
|
|
"""Item 501: a hidden probe is shown once it has filled part of the energy gate."""
|
|
if self._probe_hidden and self._energy_since_idle_wh >= (
|
|
STANDBY_REPROBE_SHOW_ENERGY_FRACTION * self._config.start_energy_threshold
|
|
):
|
|
self._probe_hidden = False
|
|
|
|
@property
|
|
def _gate_cadence(self) -> float:
|
|
"""Cadence the pause/end gates are sized from (#424/#427).
|
|
|
|
``_p95_dt`` is the 2nd-largest of the last 20 intervals by construction,
|
|
so it tracks the worst *gap* rather than the reporting rate. That is the
|
|
right input for the outage ceilings (a gap must be judged against the
|
|
worst gap we consider normal) but the wrong one for the gates below,
|
|
which multiply it by three: once a publish-on-change plug falls silent at
|
|
standby, the only intervals left are the long quiet ones, p95 collapses
|
|
onto them, and each gate becomes ~3x the silence it is supposed to be
|
|
measuring. The accumulator advances one interval per reading, so the
|
|
cycle then needs ~3 more readings - a gate that is set by, and grows
|
|
with, its own input. Measured on the #427 trace: a 297 s sensor silence
|
|
followed by a 481 s keepalive lifted the end gate from 45 s to 1455 s and
|
|
held the cycle in PAUSED for 25.5 min.
|
|
|
|
Capping p95 at a multiple of the *median* keeps the estimate robust: a
|
|
genuinely slow sensor reports slowly every time, so its median equals its
|
|
p95 and the cap never binds (a 300 s-cadence meter keeps its 900 s gate);
|
|
a fast sensor that went quiet has a small median, so the isolated holes
|
|
cannot triple the gate. Only these two gates read it - ``_p95_dt`` itself
|
|
is left alone so every outage ceiling keeps the cadence snapshot it was
|
|
tuned against (register items 213, 215).
|
|
|
|
The cap alone was not enough, and the reason is worth keeping: it is
|
|
computed over the same 20-interval window it is meant to protect. Once a
|
|
publish-on-change plug falls silent the watchdog's own 0 W keepalives are
|
|
the only readings left, so the median collapses onto the injection spacing
|
|
too and ``5 x median`` stops binding - the gate then grows with our
|
|
injection rate instead of the plug's. Fixed at the source (register item
|
|
289): ``process_reading`` no longer trains the cadence on synthetic
|
|
readings, so this window describes the sensor and nothing else.
|
|
"""
|
|
if len(self._recent_dts) < 5:
|
|
return self._p95_dt
|
|
median_dt = median_fast(self._recent_dts)
|
|
return min(self._p95_dt, GATE_CADENCE_MEDIAN_FACTOR * median_dt)
|
|
|
|
@property
|
|
def _dynamic_pause_threshold(self) -> float:
|
|
"""Calculate dynamic pause threshold based on sampling cadence."""
|
|
# User requirement: T_pause >= 3 * p95_update_interval
|
|
# Default 15s or 3 * p95
|
|
return max(15.0, 3.0 * self._gate_cadence)
|
|
|
|
@property
|
|
def _dynamic_end_threshold(self) -> float:
|
|
"""Calculate dynamic end candidate threshold."""
|
|
# Keep this generic for pause->ending transitions across all device types.
|
|
base = 3.0 * self._gate_cadence
|
|
# Ensure end threshold is at least 15s greater than pause threshold
|
|
return max(base, self._dynamic_pause_threshold + 15.0)
|
|
|
|
def _update_cadence(self, dt: float) -> None:
|
|
"""Update rolling cadence statistics."""
|
|
if dt <= 0.1:
|
|
return
|
|
self._recent_dts.append(dt)
|
|
if len(self._recent_dts) > 20:
|
|
self._recent_dts.pop(0)
|
|
|
|
# Calculate p95 if enough samples
|
|
if len(self._recent_dts) >= 5:
|
|
# Pure Python, bit-identical to np.percentile (audit PERF-07): on 20
|
|
# values NumPy's call overhead was the detector's top per-sample cost.
|
|
self._p95_dt = percentile_linear(self._recent_dts, 95)
|
|
else:
|
|
self._p95_dt = max(dt, 1.0)
|
|
|
|
def _try_profile_match(self, timestamp: datetime, force: bool = False) -> None:
|
|
"""Attempt to invoke the profile matcher if conditions are met.
|
|
|
|
Args:
|
|
timestamp: Current timestamp.
|
|
force: If True, run match immediately regardless of interval.
|
|
"""
|
|
if not self._profile_matcher:
|
|
return
|
|
if not self._power_readings:
|
|
return
|
|
|
|
# Terminal-tail match freeze (dishwashers): once we are in ENDING with a
|
|
# profile already matched and power has been sustained-quiet, the active
|
|
# cycle is over - only the passive drain/dry tail remains. Re-matching on
|
|
# the growing idle tail inflates the observed duration and drifts the
|
|
# Stage-4 duration-agreement toward a LONGER near-duplicate profile,
|
|
# flipping the label and stalling smart-termination on the ambiguity gate.
|
|
# Keep the active-phase match instead. Self-correcting: a real resume sends
|
|
# a high reading that leaves ENDING, so this guard stops applying.
|
|
if (
|
|
self._state == STATE_ENDING
|
|
and self._config.device_type == "dishwasher"
|
|
and self._matched_profile
|
|
and self._time_below_threshold >= DISHWASHER_MATCH_FREEZE_QUIET_SECONDS
|
|
):
|
|
# The freeze also stops the manager's match tick, the only place the #375
|
|
# sustained-quiet release ran: a verified pause engaged before the freeze
|
|
# was never released, and every ENDING finalize stayed blocked until the
|
|
# force stop (~8 h on a plug that keeps reporting 0 W; found by the F7
|
|
# parity harness on three TRON4R dishwasher cycles). Run the same rule here.
|
|
if self._verified_pause and not self._user_paused and self._power_readings:
|
|
pause = match_rules.decide_pause_release(
|
|
verified_pause=True,
|
|
current_matched=self._matched_profile,
|
|
current_power=float(self._power_readings[-1][1]),
|
|
stop_threshold_w=float(self._config.stop_threshold_w),
|
|
user_paused=False,
|
|
expected_duration=float(self._expected_duration or 0.0),
|
|
current_duration=(
|
|
self._power_readings[-1][0] - self._power_readings[0][0]
|
|
).total_seconds(),
|
|
time_below=self._time_below_threshold_gapfree,
|
|
program=self._matched_profile,
|
|
)
|
|
for level, msg, args in pause.log:
|
|
self._logger.log(level, msg, *args)
|
|
self._verified_pause = pause.verified_pause
|
|
return
|
|
|
|
# Terminal-tail match freeze (anti-crease, #296): once a washer/dryer with
|
|
# anti-wrinkle enabled is past its expected duration and has settled into
|
|
# the low-power tumble tail, re-matching on the growing flat tail drifts the
|
|
# label toward a LONGER near-duplicate (its expected duration grows), which
|
|
# pushes the anti-crease finalize gate out and breaks Smart Termination -
|
|
# the field failure that merges back-to-back washes. Keep the good
|
|
# pre-tail match instead. Self-correcting: a new wash's heating burst above
|
|
# anti_wrinkle_max_power leaves the tail regime, so this stops applying and
|
|
# matching re-arms for the next cycle.
|
|
if self._matched_profile and self._in_anticrease_freeze(timestamp):
|
|
return
|
|
|
|
# Rate limiting
|
|
if not force and self._last_match_time:
|
|
elapsed = (timestamp - self._last_match_time).total_seconds()
|
|
interval = float(self._config.match_interval)
|
|
if not getattr(self, "_match_committed", True):
|
|
interval /= 2.0
|
|
if elapsed < interval:
|
|
return
|
|
|
|
self._last_match_time = timestamp
|
|
|
|
# Call the matcher
|
|
try:
|
|
result = self._profile_matcher(self._match_readings())
|
|
# If synchronous result returned, process it.
|
|
# If None returned (async offload), the matcher is responsible for
|
|
# calling update_match later.
|
|
if result:
|
|
self.update_match(result)
|
|
|
|
except Exception as e: # pylint: disable=broad-exception-caught
|
|
self._logger.debug("Profile match failed: %s", e)
|
|
|
|
# Maximum reasonable cycle duration accepted by the detector. Anything
|
|
# longer is rejected as corrupted data and replaced with the
|
|
# _SANITIZE_INVALID_SENTINEL so downstream gates fall through to the
|
|
# unmatched / no-expected-duration path.
|
|
_SANITIZE_MAX_EXPECTED_DURATION = 6 * 3600.0 # 6 hours
|
|
_SANITIZE_INVALID_SENTINEL = 0.0 # 0 == "no valid expected_duration"
|
|
|
|
def _sanitize_expected_duration(
|
|
self, raw: Any, *, source: str = "update_match"
|
|
) -> float:
|
|
"""Coerce ``raw`` into a finite float in (0, 6h] or return 0.0.
|
|
|
|
The class invariant is that ``self._expected_duration`` is either a
|
|
finite, strictly positive float ≤ 6 hours, or 0.0 meaning "no valid
|
|
expected duration". Every code path that assigns ``_expected_duration``
|
|
(live profile-match callbacks AND restored snapshots) routes through
|
|
this helper so the gates in STATE_ENDING and ``_should_defer_finish``
|
|
can trust the value without re-validating.
|
|
|
|
Emits a DEBUG log line distinguishing the rejection reason - the
|
|
``<= 0`` and ``> 6h`` markers are part of issue #197's regression
|
|
contract and tests assert on them.
|
|
"""
|
|
try:
|
|
value = float(raw)
|
|
except (TypeError, ValueError, OverflowError):
|
|
self._logger.debug(
|
|
"%s: invalid raw_expected_duration %r, defaulting to 0.0",
|
|
source, raw,
|
|
)
|
|
return self._SANITIZE_INVALID_SENTINEL
|
|
if not math.isfinite(value):
|
|
self._logger.debug(
|
|
"%s: invalid raw_expected_duration %r, defaulting to 0.0",
|
|
source, raw,
|
|
)
|
|
return self._SANITIZE_INVALID_SENTINEL
|
|
if value <= 0:
|
|
self._logger.debug(
|
|
"%s: invalid raw_expected_duration %r (<= 0), defaulting to 0.0",
|
|
source, raw,
|
|
)
|
|
return self._SANITIZE_INVALID_SENTINEL
|
|
if value > self._SANITIZE_MAX_EXPECTED_DURATION:
|
|
self._logger.debug(
|
|
"%s: invalid raw_expected_duration %r (> 6h), defaulting to 0.0",
|
|
source, raw,
|
|
)
|
|
return self._SANITIZE_INVALID_SENTINEL
|
|
return value
|
|
|
|
@staticmethod
|
|
def _sanitize_terminal_quiet(raw: Any) -> float | None:
|
|
"""Coerce a measured post-activity quiet span into a finite, non-negative
|
|
float, else None (register item 297).
|
|
|
|
Same discipline as the two siblings: None means "no opinion", and for a
|
|
DISHWASHER ``_keep_tail_cap`` then behaves exactly as it did before this
|
|
element existed. Other device types never reach that branch - they are
|
|
capped at ``_last_active_time`` higher up - so None changes nothing for
|
|
them either way. Bounded above by ``TERMINAL_QUIET_CAP_S`` so a corrupted
|
|
or hand-edited value cannot license an unbounded tail - the one thing
|
|
this field exists to prevent.
|
|
"""
|
|
if raw is None:
|
|
return None
|
|
try:
|
|
value = float(raw)
|
|
except (TypeError, ValueError, OverflowError):
|
|
return None
|
|
if not math.isfinite(value) or value < 0:
|
|
return None
|
|
return min(value, TERMINAL_QUIET_CAP_S)
|
|
|
|
@staticmethod
|
|
def _sanitize_positive(raw: Any) -> float | None:
|
|
"""A finite positive float from a snapshot, else None."""
|
|
if raw is None or isinstance(raw, bool):
|
|
return None
|
|
try:
|
|
value = float(raw)
|
|
except (TypeError, ValueError, OverflowError):
|
|
return None
|
|
return value if math.isfinite(value) and value > 0.0 else None
|
|
|
|
@staticmethod
|
|
def _sanitize_trusted_min(raw: Any) -> float | None:
|
|
"""Coerce element 13 to a positive duration or None ("nothing vouched")."""
|
|
if raw is None or isinstance(raw, bool):
|
|
return None
|
|
try:
|
|
value = float(raw)
|
|
except (TypeError, ValueError, OverflowError):
|
|
return None
|
|
if not math.isfinite(value) or not 0.0 < value <= CycleDetector._SANITIZE_MAX_EXPECTED_DURATION:
|
|
return None
|
|
return value
|
|
|
|
@staticmethod
|
|
def _sanitize_pause_catalogue(
|
|
raw: Any,
|
|
) -> tuple[int, tuple[tuple[float, float], ...]] | None:
|
|
"""Coerce element 14 to ``(traced, ((fraction, seconds), ...))`` or None."""
|
|
try:
|
|
traced, pauses = raw
|
|
out = tuple(
|
|
(float(f), float(d)) for f, d in pauses
|
|
if math.isfinite(float(f)) and math.isfinite(float(d)) and float(d) >= 0
|
|
)
|
|
return (int(traced), out) if int(traced) > 0 else None
|
|
except (TypeError, ValueError, OverflowError):
|
|
return None
|
|
|
|
def _hazard_wait(self, timestamp: datetime, base: float) -> float:
|
|
"""The ENDING fallback wait the matched profile's own pauses justify.
|
|
|
|
Audit DETECT-16: "no soak left after 1.05x expected" generalised to "no
|
|
soak this programme has ever shown from this position": END_GATE_HAZARD_
|
|
MARGIN x the longest resumed pause its traced evidence began at or after
|
|
this quiet's start (less a slack), clamped to [off_delay, base]. Shorten-
|
|
only, unambiguous matches only, and only with END_GATE_HAZARD_MIN_CYCLES
|
|
traced cycles; otherwise ``base``. Measured (prototype, 292 cycles): washer
|
|
median lag 16.17 -> 12.50 min, early ends 0 -> 0, splits unchanged.
|
|
"""
|
|
return self._hazard_wait_for(
|
|
timestamp, base, self._matched_pause_catalogue,
|
|
self._expected_duration, self._match_ambiguous,
|
|
)
|
|
|
|
def _hazard_wait_for(
|
|
self, timestamp: datetime, base: float, cat: Any, expected: float, ambiguous: bool
|
|
) -> float:
|
|
""":meth:`_hazard_wait` for a match with these values (register item 469b)."""
|
|
if (
|
|
cat is None
|
|
or cat[0] < END_GATE_HAZARD_MIN_CYCLES
|
|
or not self._matched_profile
|
|
or expected <= 0
|
|
or self._current_cycle_start is None
|
|
or ambiguous
|
|
):
|
|
return base
|
|
elapsed = self._gate_elapsed_s(timestamp) # programme time (item 511)
|
|
# Where the quiet run began. An outage inside it is still part of the run
|
|
# (item 266), or the outage would read as progress and drop the very pause
|
|
# being waited out from `later`.
|
|
quiet_run = self._time_below_threshold + self._time_below_unobserved
|
|
position = max(0.0, (elapsed - quiet_run) / expected)
|
|
later = [d for f, d in cat[1] if f >= position - END_GATE_HAZARD_POSITION_SLACK]
|
|
need = END_GATE_HAZARD_MARGIN * max(later) if later else 0.0
|
|
return max(float(self._config.off_delay), min(float(base), need))
|
|
|
|
@staticmethod
|
|
def _sanitize_longest_candidate(raw: Any) -> float:
|
|
"""Coerce the longest plausible candidate duration to a usable bound.
|
|
|
|
Unlike the three siblings above, "no opinion" is ``0.0`` rather than
|
|
``None``: the ENDING gate reads a non-positive bound as "no information"
|
|
and keeps the old refusal, which is the safe direction. Not routed
|
|
through ``_sanitize_expected_duration`` because 0.0 is legitimate here
|
|
and that helper logs it as invalid.
|
|
"""
|
|
try:
|
|
value = float(raw or 0.0)
|
|
except (TypeError, ValueError, OverflowError):
|
|
return 0.0
|
|
if not math.isfinite(value):
|
|
return 0.0
|
|
return value if 0.0 < value <= CycleDetector._SANITIZE_MAX_EXPECTED_DURATION else 0.0
|
|
|
|
@staticmethod
|
|
def _sanitize_tail_power(raw: Any) -> float | None:
|
|
"""Coerce ``raw`` into a finite, positive float, else None (#364).
|
|
|
|
None means "no opinion": ``_smart_term_power_plausible`` then leaves both
|
|
Smart-Termination paths exactly as they behaved before the guard existed.
|
|
"""
|
|
if raw is None:
|
|
return None
|
|
try:
|
|
value = float(raw)
|
|
except (TypeError, ValueError, OverflowError):
|
|
# OverflowError alongside the type errors: `json` keeps an integer
|
|
# literal of any length as an unbounded `int`, and `float()` on one
|
|
# raises rather than returning `inf`, so the non-finite filter below is
|
|
# never reached. Both callers are the reason it matters - the restored
|
|
# state snapshot is hand-editable `.storage` JSON, and
|
|
# `restore_state_snapshot`'s one broad `except` answers a raise with
|
|
# `self.reset()`, which discards the WHOLE restored cycle rather than
|
|
# this one field. "No opinion" is the documented contract; a full reset
|
|
# is not.
|
|
return None
|
|
if not math.isfinite(value) or value <= 0:
|
|
return None
|
|
return value
|
|
|
|
@staticmethod
|
|
def _sanitize_terminal_high(
|
|
raw: Any,
|
|
) -> (
|
|
tuple[float, float]
|
|
| tuple[float, float, float]
|
|
| tuple[float, float, float, float]
|
|
| None
|
|
):
|
|
"""Coerce ``raw`` into a ``(start_frac, seconds)`` pair, a
|
|
``(start_frac, seconds, start_offset_s)`` triple, or a
|
|
``(start_frac, seconds, start_offset_s, ceiling_w)`` quad, else None (#399).
|
|
|
|
The fourth element is the watts the block was MEASURED against (register
|
|
item 351). Anti-crease measures against ``anti_wrinkle_max_power`` and
|
|
sends a triple; the standby-band path measures against a share of the
|
|
cycle's own peak and must say so, because the live counter has to count
|
|
seconds above the same bar or the two halves compare different things.
|
|
|
|
None means "no opinion", which leaves ``_anticrease_spin_pending`` inert and
|
|
the anti-crease finalise exactly as it behaved before the guard existed.
|
|
|
|
The arity is PRESERVED rather than normalised (register item 196). A
|
|
two-element payload is what a pre-196 state snapshot, an older caller and
|
|
most tests supply, and it has to keep meaning "no absolute offset, fall back
|
|
to ``start_frac x expected``" - handing it a fabricated 0.0 offset would make
|
|
the guard scan the whole cycle. A malformed third element degrades to the
|
|
pair for the same reason the whole method returns None on garbage: it must
|
|
never be able to disarm a guard that would otherwise arm.
|
|
"""
|
|
if raw is None:
|
|
return None
|
|
# A str/bytes is iterable, so `list("11")` is `["1", "1"]` and sanitizes to
|
|
# (1.0, 1.0) - a scalar string silently ARMING the guard off a malformed
|
|
# snapshot, which is the one direction this method promises never to go.
|
|
# Rejected before the iteration so it takes the documented garbage path.
|
|
if isinstance(raw, (str, bytes, bytearray)):
|
|
return None
|
|
try:
|
|
values = list(raw)
|
|
except TypeError:
|
|
return None
|
|
if len(values) not in (2, 3, 4):
|
|
return None
|
|
try:
|
|
start_frac = float(values[0])
|
|
seconds = float(values[1])
|
|
except (TypeError, ValueError, OverflowError):
|
|
# Same reason as _sanitize_tail_power above: an unbounded int raises out
|
|
# of float(), and a raise here costs the whole restored cycle state.
|
|
return None
|
|
if not math.isfinite(start_frac) or not math.isfinite(seconds):
|
|
return None
|
|
if not 0.0 <= start_frac <= 1.0 or seconds <= 0:
|
|
return None
|
|
if len(values) == 2:
|
|
return (start_frac, seconds)
|
|
try:
|
|
start_offset = float(values[2])
|
|
except (TypeError, ValueError, OverflowError):
|
|
return (start_frac, seconds)
|
|
if not math.isfinite(start_offset) or start_offset < 0:
|
|
return (start_frac, seconds)
|
|
if len(values) == 3:
|
|
return (start_frac, seconds, start_offset)
|
|
try:
|
|
ceiling = float(values[3])
|
|
except (TypeError, ValueError, OverflowError):
|
|
return (start_frac, seconds, start_offset)
|
|
# A non-positive or non-finite ceiling degrades to the triple rather than
|
|
# to None: the triple still arms the guard against
|
|
# ``anti_wrinkle_max_power``, and this method must never be able to
|
|
# DISARM a guard that would otherwise arm.
|
|
if not math.isfinite(ceiling) or ceiling <= 0:
|
|
return (start_frac, seconds, start_offset)
|
|
return (start_frac, seconds, start_offset, ceiling)
|
|
|
|
def _trailing_mean_power(self, timestamp: datetime, window_s: float) -> float | None:
|
|
"""Time-weighted mean power over the trailing ``window_s``, or None when
|
|
there are too few samples to judge.
|
|
|
|
Time-weighted (not a plain sample mean) so an irregular reporting cadence -
|
|
a plug that only pushes on change, so quiet stretches are sparse - cannot
|
|
bias the result toward whichever regime happened to sample more often.
|
|
"""
|
|
window: list[tuple[datetime, float]] = []
|
|
for ts, power in reversed(self._power_readings):
|
|
if (timestamp - ts).total_seconds() > window_s:
|
|
break
|
|
window.append((ts, float(power)))
|
|
window.reverse()
|
|
# Explicit gap handling (energy-integration rule): a reading held across an
|
|
# unobserved outage would dominate the trapezoid - e.g. a stale 2000 W sample
|
|
# 290 s before a 284 s dropout, then 5 W, integrates to ~1000 W though every
|
|
# observed recent reading is 5 W, wrongly blocking termination. So drop
|
|
# everything up to and including the most recent outage-sized gap and judge
|
|
# only the clean contiguous tail, the same ceiling the standby / anti-crease
|
|
# window scans reject a holed window with.
|
|
if len(window) >= 2:
|
|
# O(1) ceiling from the maintained p95 cadence (mirrors energy_gap_threshold_s
|
|
# = clip(10x cadence, 60, 3600)), NOT _outage_threshold_s() which rebuilds a
|
|
# NumPy array from every reading - this runs on the per-reading ENDING /
|
|
# anti-crease path, same reasoning as the gap-free tally at L1006.
|
|
max_gap = min(3600.0, max(60.0, 10.0 * self._prior_p95_dt))
|
|
cut = 0
|
|
for i in range(1, len(window)):
|
|
if (window[i][0] - window[i - 1][0]).total_seconds() > max_gap:
|
|
cut = i
|
|
window = window[cut:]
|
|
if len(window) < SMART_TERM_TAIL_MIN_POINTS:
|
|
return None
|
|
span = (window[-1][0] - window[0][0]).total_seconds()
|
|
if span <= 0:
|
|
return None
|
|
energy = 0.0
|
|
for (t0, p0), (t1, p1) in zip(window, window[1:]):
|
|
energy += (p0 + p1) / 2.0 * (t1 - t0).total_seconds()
|
|
return energy / span
|
|
|
|
def _smart_term_power_plausible(self, timestamp: datetime) -> bool:
|
|
"""Whether the appliance looks like it is actually FINISHING (#364).
|
|
|
|
Both Smart-Termination paths fire at ``elapsed >= 0.98 * expected`` and
|
|
neither asks whether the machine is still working. When the matcher has
|
|
locked onto a shorter look-alike profile that anchor lands mid-wash, the
|
|
cycle is cut in half and the remainder is recorded as a second cycle.
|
|
|
|
The test: compare the trailing mean power against what the matched profile
|
|
itself draws at its own end. Drawing several times that level is proof we
|
|
are not at the end of anything - whatever the clock says. Unlike the
|
|
prefix-landscape guard this needs no longer profile to exist in the pool,
|
|
so it also covers the reported case where the programme actually running
|
|
was never trained.
|
|
|
|
Shorten-only and fail-open: any missing input returns True, leaving
|
|
behaviour identical to before the guard. A False can only ever *block* an
|
|
early finish - the power-based fallback timeout still ends the cycle.
|
|
"""
|
|
tail_power = self._matched_tail_power
|
|
if tail_power is None or tail_power <= 0:
|
|
return True
|
|
mean_power = self._trailing_mean_power(timestamp, self._tail_window_s())
|
|
if mean_power is None:
|
|
return True
|
|
return mean_power <= tail_power * SMART_TERM_TAIL_MAX_RATIO
|
|
|
|
def _tail_window_s(self) -> float:
|
|
"""Trailing window that covers the same FRACTION of the run as the profile
|
|
tail it is compared against.
|
|
|
|
A fixed window is not comparable across programme lengths: 300 s is 4% of a
|
|
cotton wash but a third of a 15-minute spin-and-drain, whose trailing mean
|
|
would then be the spin itself while its profile tail is the quiet moment
|
|
after the pump stops. Measured on the full corpus, making this proportional
|
|
is strictly better at every threshold.
|
|
"""
|
|
expected = self._expected_duration
|
|
if expected <= 0:
|
|
return SMART_TERM_TAIL_WINDOW_S
|
|
return min(
|
|
SMART_TERM_TAIL_WINDOW_S,
|
|
max(SMART_TERM_TAIL_WINDOW_MIN_S, expected * SMART_TERM_TAIL_WINDOW_FRAC),
|
|
)
|
|
|
|
def update_match(self, result: tuple[Any, ...] | list[Any] | Any) -> None: # type: ignore[misc]
|
|
"""Process a match result (synchronously).
|
|
|
|
Can be called by the matcher callback directly or asynchronously.
|
|
"""
|
|
# Terminal-tail match freeze (anti-crease, #296). This is the single sink
|
|
# for ALL match updates - the detector's own _try_profile_match AND the
|
|
# manager's async 5-min matcher (manager.py calls update_match directly).
|
|
# Once a washer/dryer with anti-wrinkle enabled is past its expected
|
|
# duration and has settled into the low-power tumble tail, re-matching on
|
|
# the growing flat tail drifts the label toward a LONGER near-duplicate (or
|
|
# flips it ambiguous), which pushes out expected_duration and would block
|
|
# the anti-crease finalize - the field failure that merges back-to-back
|
|
# washes. Keep the good pre-tail match instead. Self-correcting: a new
|
|
# wash's heating burst above anti_wrinkle_max_power leaves the tail regime,
|
|
# so this stops applying and matching re-arms for the next cycle.
|
|
if (
|
|
self._matched_profile
|
|
and self._power_readings
|
|
and self._in_anticrease_freeze(self._power_readings[-1][0])
|
|
):
|
|
return
|
|
if isinstance(result, MatchContext):
|
|
result = result.as_sequence()
|
|
# Register item 469(b): an ambiguous match in ENDING may not make the end
|
|
# later (see `ambiguous_ending_match_defers`). Like the freezes above, the
|
|
# detector keeps the match it has; a revoke still applies.
|
|
if (
|
|
match_rules.HOLD_AMBIGUOUS_IN_ENDING
|
|
and isinstance(result, (list, tuple))
|
|
and len(result) >= 6
|
|
and result[0] is not None
|
|
and bool(result[5])
|
|
and not bool(result[4])
|
|
and self.ambiguous_ending_match_defers(
|
|
result[2], result[1], result[11] if len(result) >= 12 else 0.0,
|
|
result[13] if len(result) >= 14 else None,
|
|
)
|
|
):
|
|
self._logger.debug(
|
|
"ENDING: ambiguous match %s not applied, %s kept (item 469)",
|
|
result[0], self._matched_profile,
|
|
)
|
|
return
|
|
# Unpack 5 elements (or 4 for backward compatibility if needed, but wrapper is updated)
|
|
# wrapper returns (name, confidence, duration, phase, is_mismatch)
|
|
# Or MatchResult object if refactored, but currently wrapper returns tuple.
|
|
|
|
# Register item 498: what a revoke below discards, kept as the evidence
|
|
# that bounds a verified pause it leaves behind.
|
|
revoked_from = self._matched_profile
|
|
revoked_catalogue = self._matched_pause_catalogue
|
|
|
|
is_match_mismatch = False
|
|
match_name: str | None = None
|
|
phase_name: str | None = None
|
|
confidence: float = 0.0
|
|
expected_duration: float = 0.0
|
|
ambiguous: bool = False
|
|
|
|
if isinstance(result, (list, tuple)): # type: ignore[misc]
|
|
result_seq = cast(tuple[Any, ...] | list[Any], result)
|
|
# Optional 6th element: whether the live match is ambiguous
|
|
# (top-1 vs top-2 within MATCH_AMBIGUITY_MARGIN). Used to gate the
|
|
# predictive Smart Termination below.
|
|
if len(result_seq) >= 6:
|
|
ambiguous = bool(result_seq[5])
|
|
if len(result_seq) >= 5:
|
|
(
|
|
raw_name,
|
|
raw_confidence,
|
|
raw_expected_duration,
|
|
raw_phase_name,
|
|
raw_mismatch,
|
|
) = result_seq[:5]
|
|
match_name = str(raw_name) if raw_name is not None else None
|
|
try:
|
|
confidence = float(raw_confidence)
|
|
if not math.isfinite(confidence):
|
|
confidence = 0.0
|
|
self._logger.debug("update_match: invalid raw_confidence %r, defaulting to 0.0", raw_confidence)
|
|
except (TypeError, ValueError, OverflowError):
|
|
confidence = 0.0
|
|
self._logger.debug("update_match: invalid raw_confidence %r, defaulting to 0.0", raw_confidence)
|
|
expected_duration = self._sanitize_expected_duration(
|
|
raw_expected_duration, source="update_match"
|
|
)
|
|
phase_name = str(raw_phase_name) if raw_phase_name is not None else None
|
|
is_match_mismatch = raw_mismatch if isinstance(raw_mismatch, bool) else bool(raw_mismatch)
|
|
else:
|
|
# Fallback for old signature
|
|
if len(result_seq) >= 4:
|
|
(
|
|
raw_name,
|
|
raw_confidence,
|
|
raw_expected_duration,
|
|
raw_phase_name,
|
|
) = result_seq[:4]
|
|
match_name = str(raw_name) if raw_name is not None else None
|
|
try:
|
|
confidence = float(raw_confidence)
|
|
if not math.isfinite(confidence):
|
|
confidence = 0.0
|
|
self._logger.debug("update_match: invalid raw_confidence %r, defaulting to 0.0", raw_confidence)
|
|
except (TypeError, ValueError, OverflowError):
|
|
confidence = 0.0
|
|
self._logger.debug("update_match: invalid raw_confidence %r, defaulting to 0.0", raw_confidence)
|
|
expected_duration = self._sanitize_expected_duration(
|
|
raw_expected_duration, source="update_match"
|
|
)
|
|
phase_name = (
|
|
str(raw_phase_name) if raw_phase_name is not None else None
|
|
)
|
|
is_match_mismatch = False
|
|
|
|
# Store confidence + ambiguity for Smart Termination checks
|
|
self._last_match_confidence = confidence or 0.0
|
|
self._match_ambiguous = ambiguous
|
|
# Element 8: the #288 full-shape term. Element 7 (the removed #364
|
|
# prefix-fit flag) is read only as its fallback for a 7-element legacy
|
|
# tuple, which carried the #288 verdict there before #364 split it out.
|
|
self._match_prefix_ambiguous_full_shape = (
|
|
bool(result_seq[7]) if len(result_seq) >= 8
|
|
else bool(result_seq[6]) if len(result_seq) >= 7
|
|
else False
|
|
)
|
|
# Element 9 (#364): the matched profile's own tail power level. Absent
|
|
# or non-finite leaves the power-plausibility guard inert - so a shorter
|
|
# tuple (Playground, older callers, most tests) must CLEAR it, not keep
|
|
# the previous match's value: retaining it would let the guard compare
|
|
# the live tail against the wrong profile and block a valid termination.
|
|
self._matched_tail_power = (
|
|
self._sanitize_tail_power(result_seq[8]) if len(result_seq) >= 9 else None
|
|
)
|
|
# Element 10 (#399): the matched profile's own terminal high-power
|
|
# block. Cleared by a shorter tuple for the same reason as element 9 -
|
|
# keeping the previous profile's block would make the anti-crease guard
|
|
# wait for a spin the newly-matched program does not have.
|
|
self._matched_terminal_high = (
|
|
self._sanitize_terminal_high(result_seq[9]) if len(result_seq) >= 10 else None
|
|
)
|
|
# Element 11 (register item 297): cleared by a shorter tuple for the same reason as
|
|
# elements 9 and 10 - a newly matched programme must not inherit the
|
|
# previous one's tail.
|
|
self._matched_terminal_quiet_s = (
|
|
self._sanitize_terminal_quiet(result_seq[10])
|
|
if len(result_seq) >= 11
|
|
else None
|
|
)
|
|
# Element 12: longest plausible candidate duration (see the attribute's
|
|
# own comment). A shorter tuple clears it, like elements 9-11, so a
|
|
# stale value can never license a shortening for a different match.
|
|
# Coerced quietly, not through _sanitize_expected_duration: 0.0 is a
|
|
# legitimate "no candidate durations to compare" here, and that helper
|
|
# logs it as invalid.
|
|
self._longest_candidate_duration = self._sanitize_longest_candidate(
|
|
result_seq[11] if len(result_seq) >= 12 else 0.0
|
|
)
|
|
# Element 13: cleared by a shorter tuple, like elements 9-12.
|
|
self._matched_trusted_min_s = self._sanitize_trusted_min(
|
|
result_seq[12] if len(result_seq) >= 13 else None
|
|
)
|
|
# Element 14: cleared by a shorter tuple, like elements 9-13.
|
|
self._matched_pause_catalogue = self._sanitize_pause_catalogue(
|
|
result_seq[13] if len(result_seq) >= 14 else None
|
|
)
|
|
# Element 15 (#452): likewise. A run already in progress keeps the
|
|
# match it began under (`_stall_match`).
|
|
self._matched_stall_catalogue = self._sanitize_stall_catalogue(
|
|
result_seq[14] if len(result_seq) >= 15 else None
|
|
)
|
|
self._stall_now = None # the current match is re-read for a run in progress
|
|
else:
|
|
# Assume MatchResult object or similar (future proofing)
|
|
# But for now wrapper returns tuple
|
|
return
|
|
|
|
if is_match_mismatch and self._matched_profile:
|
|
# Confident non-match - revert to detecting if previously matched.
|
|
# The expected duration goes with it, as in reset(): the duration-
|
|
# anchored hard finalize, the dishwasher end-spike arm gate, the
|
|
# keep-tail cap and the manager's zombie killer read it without
|
|
# checking `_matched_profile`, so a revoked programme's length kept
|
|
# steering an unmatched cycle.
|
|
self._matched_profile = None
|
|
self._expected_duration = 0.0
|
|
self._match_ambiguous = False
|
|
self._match_prefix_ambiguous_full_shape = False
|
|
self._matched_tail_power = None
|
|
self._matched_terminal_high = None
|
|
self._matched_terminal_quiet_s = None
|
|
self._matched_trusted_min_s = None
|
|
self._matched_pause_catalogue = None
|
|
self._matched_stall_catalogue = None
|
|
|
|
elif match_name:
|
|
# If sanitization rejected the expected_duration, treat the match
|
|
# as invalid: setting _matched_profile while _expected_duration is
|
|
# the 0.0 sentinel would let Smart Termination fire on the
|
|
# `current_duration >= 0` always-true comparison. Drop both so
|
|
# the cycle stays in detecting/unmatched mode.
|
|
if expected_duration == self._SANITIZE_INVALID_SENTINEL:
|
|
self._logger.debug(
|
|
"update_match: match %r ignored - expected_duration "
|
|
"sanitized to invalid sentinel; treating as unmatched",
|
|
match_name,
|
|
)
|
|
self._matched_profile = None
|
|
self._expected_duration = self._SANITIZE_INVALID_SENTINEL
|
|
else:
|
|
self._matched_profile = match_name
|
|
# Sub-state can be set from phase_name if available
|
|
if phase_name:
|
|
self._sub_state = phase_name
|
|
# Wrapper provides it
|
|
self._expected_duration = expected_duration
|
|
|
|
if revoked_from and not self._matched_profile and revoked_catalogue is not None:
|
|
# Either path above that drops the match (a revoke, or a named match
|
|
# with an unusable duration). Max over the cycle's revokes.
|
|
self._orphaned_pause_evidence_s = max(
|
|
self._orphaned_pause_evidence_s,
|
|
max((d for _f, d in revoked_catalogue[1]), default=0.0),
|
|
)
|
|
|
|
def _release_orphaned_pause(self) -> None:
|
|
"""Release a verified pause a revoked match left behind (register item 498).
|
|
|
|
After a revoke nothing could clear an automatic verified pause but high
|
|
power: the 95%-of-span release needs the match, the #375 release its
|
|
expected duration. So it held every ENDING finisher until the 8 h force stop
|
|
(a plug reporting 0 W) or the watchdog's 4.5 h silence limit (a silent plug).
|
|
The bounded wait is :func:`match_rules.orphaned_pause_wait_s`; a user pause
|
|
is never released. Called on every ENDING reading (watchdog keepalives and
|
|
the Playground's replay included), after this reading's match tick.
|
|
"""
|
|
if not self._verified_pause or self._user_paused or self._matched_profile:
|
|
return
|
|
longest = self._orphaned_pause_evidence_s
|
|
pause = match_rules.decide_orphaned_pause_release(
|
|
verified_pause=True,
|
|
user_paused=False,
|
|
current_matched=None,
|
|
time_below=self._time_below_threshold_gapfree,
|
|
wait_s=match_rules.orphaned_pause_wait_s(
|
|
off_delay=self._config.off_delay,
|
|
min_off_gap=self._config.min_off_gap,
|
|
longest_pause_s=longest,
|
|
),
|
|
longest_pause_s=longest,
|
|
)
|
|
for level, msg, args in pause.log:
|
|
self._logger.log(level, msg, *args)
|
|
self._verified_pause = pause.verified_pause
|
|
|
|
def set_verified_pause(self, verified: bool) -> None:
|
|
"""Set or clear the verified pause flag."""
|
|
self._verified_pause = verified
|
|
|
|
def set_match_committed(self, committed: bool) -> None:
|
|
"""Mirror whether the caller has committed a program this cycle.
|
|
|
|
Until it has, matches run at half ``match_interval`` (audit LIVE-17): at
|
|
the shipped 300 s, 12 resampled points on a 30 s plug need 330 s, so the
|
|
300 s try was empty too and the first commit came a median 25 min in.
|
|
Measured: right programme shown 56.2 -> 61.0% of cycle time, never
|
|
committed 8 -> 4 of 247, +19% matcher CPU. Persistence stays a match
|
|
count, as in the measured arm.
|
|
"""
|
|
self._match_committed = bool(committed)
|
|
|
|
def set_user_paused(self, paused: bool, timestamp: datetime | None = None) -> None:
|
|
"""Mirror the user's Pause Cycle state (set by the manager).
|
|
|
|
Kept apart from ``_verified_pause`` on purpose: the envelope auto-pause
|
|
writes that one too, and freezing Smart Termination on it would re-open
|
|
the #375 hang class. Only a USER pause blocks Smart Termination.
|
|
|
|
Item 514: a resume banks the pause (from ``timestamp``, now by default)
|
|
as a span the gates' programme time leaves out. A pause start restored
|
|
from a snapshot is kept when the manager re-asserts the pause.
|
|
"""
|
|
paused = bool(paused)
|
|
now = dt_util.as_utc(timestamp) if timestamp is not None else utc_now()
|
|
if paused and not self._user_paused:
|
|
if self._user_pause_since is None:
|
|
self._user_pause_since = now
|
|
elif not paused:
|
|
since, start = self._user_pause_since, self._current_cycle_start
|
|
self._user_pause_since = None
|
|
if self._user_paused and since is not None and start is not None:
|
|
begin = max(since, start)
|
|
length = (now - begin).total_seconds()
|
|
if length > 0:
|
|
self._user_pause_spans = _merge_spans([
|
|
*self._user_pause_spans,
|
|
((begin - start).total_seconds(), length),
|
|
])
|
|
self._user_paused = paused
|
|
|
|
def mark_sensor_unavailable(self, timestamp: datetime) -> None:
|
|
"""Start a sensor outage: the power sensor has no usable value (item 266).
|
|
|
|
Called by the manager when the sensor's state becomes unavailable, unknown
|
|
or non-numeric. Until the next real reading nothing is observed, so that
|
|
span is not credited as quiet, and a watchdog keepalive inside it takes no
|
|
decision (see ``process_reading``). Idempotent: an ongoing outage keeps
|
|
its first start.
|
|
"""
|
|
if self._sensor_outage_since is None:
|
|
self._sensor_outage_since = dt_util.as_utc(timestamp)
|
|
in_cycle = self._state in (
|
|
STATE_STARTING, STATE_RUNNING, STATE_PAUSED, STATE_ENDING
|
|
)
|
|
self._logger.log(
|
|
logging.INFO if in_cycle else logging.DEBUG,
|
|
"Power sensor unavailable from %s (state %s): quiet time is not "
|
|
"credited until it reports again",
|
|
self._sensor_outage_since,
|
|
self._state,
|
|
)
|
|
|
|
def reset(
|
|
self, target_state: str = STATE_OFF, timestamp: datetime | None = None
|
|
) -> None:
|
|
"""Force reset the detector state to target state.
|
|
|
|
``timestamp`` is the reading that caused the reset. The state entered here
|
|
is timed from it (ANTI_WRINKLE's 2 h exit reads ``_state_enter_time``), so a
|
|
replay with historical timestamps must pass it: stamping the host clock
|
|
left a replayed dryer in ANTI_WRINKLE forever (audit DETECT-10).
|
|
"""
|
|
# Item 515: the manager's Off expiry of a terminal state is a display change,
|
|
# not a fresh start. Since false starts return to the terminal state, the
|
|
# standby's re-probe flag (item 501) and the probes' pre-roll readings live
|
|
# there now, as they did in OFF, and carry into it (the cycle end emptied
|
|
# the buffer, so it holds nothing of that cycle).
|
|
idle_carry = (
|
|
TERMINAL_PROBE_RETURNS
|
|
and target_state == STATE_OFF
|
|
and self._state in (STATE_FINISHED, STATE_INTERRUPTED, STATE_FORCE_STOPPED)
|
|
)
|
|
carried = (self._standby_reprobe, self._preroll_buffer) if idle_carry else None
|
|
self._transition_to(
|
|
target_state,
|
|
dt_util.as_utc(timestamp) if timestamp is not None else utc_now(),
|
|
)
|
|
self._power_readings = []
|
|
# #430: the pre-roll buffer is pre-CYCLE context, never cross-cycle. A
|
|
# reset that left it populated would let the previous cycle's tail be
|
|
# spliced into the front of the next cycle's curve.
|
|
self._preroll_buffer = []
|
|
self._current_cycle_start = None
|
|
self._last_active_time = None
|
|
self._cycle_max_power = 0.0
|
|
self._energy_since_idle_wh = 0.0
|
|
self._time_above_threshold = 0.0
|
|
# Only reset time_below_threshold if not transitioning to ANTI_WRINKLE
|
|
# (ANTI_WRINKLE needs to track idle time to determine true-off)
|
|
if target_state != STATE_ANTI_WRINKLE:
|
|
self._time_below_threshold = 0.0
|
|
self._time_below_threshold_gapfree = 0.0
|
|
self._time_below_unobserved = 0.0
|
|
self._last_match_time = None
|
|
self._match_committed = False
|
|
self._matched_profile = None
|
|
# Clear stale match state so the next cycle starts with clean defaults.
|
|
# _expected_duration left at 0 tells the dishwasher end-spike gate that
|
|
# no profile is matched yet; stale non-zero would mis-gate the spike check.
|
|
self._expected_duration = 0.0
|
|
self._last_match_confidence = 0.0
|
|
self._match_ambiguous = False
|
|
self._match_prefix_ambiguous_full_shape = False
|
|
self._matched_tail_power = None
|
|
self._matched_terminal_high = None
|
|
self._matched_terminal_quiet_s = None
|
|
self._matched_trusted_min_s = None
|
|
self._matched_pause_catalogue = None
|
|
self._matched_stall_catalogue = None
|
|
self._orphaned_pause_evidence_s = 0.0
|
|
# Element 12 belongs with them: its own comment claims a stale value can
|
|
# never license a shortening for a different match, and that was only
|
|
# true of the tuple path. Left here across a reset, a small positive
|
|
# bound survives into the next cycle, where it is neither greater than
|
|
# `_expected_duration` (so the bar is not raised) nor <= 0 (so the "no
|
|
# information" refusal does not fire) - the one combination that lets an
|
|
# ambiguous match shorten `effective_off_delay` on no evidence.
|
|
self._longest_candidate_duration = 0.0
|
|
self._anticrease_spin_wait_logged = False
|
|
# Per-cycle diagnostic throttle (#346): the "Smart Termination not applied"
|
|
# line only logs when the reason CHANGES. Carrying the previous cycle's
|
|
# reason across a reset swallows the new cycle's very first diagnostic
|
|
# whenever it happens to be blocked for the same reason.
|
|
self._last_smart_term_block_reason = None
|
|
self._ignore_power_until_idle = False # Reset lockout
|
|
self._lockout_high_seconds = 0.0
|
|
# Clear the verified-pause flag so it can't leak into the next cycle (B6):
|
|
# a stale True would make an early low-power dip look like a verified pause
|
|
# before the first live match of the new cycle runs.
|
|
self._verified_pause = False
|
|
self._anti_wrinkle_candidate_start = None
|
|
self._anti_wrinkle_candidate_peak = 0.0
|
|
self._anti_wrinkle_candidate_start_power = 0.0
|
|
self._anticrease_tail_floor_w = None
|
|
# Reset idle time tracker for anti-wrinkle
|
|
self._anti_wrinkle_idle_time = 0.0
|
|
# Reset delayed-start tracking
|
|
self._delay_band_seconds = 0.0
|
|
self._delay_band_peak = 0.0
|
|
self._delay_wait_true_off_seconds = 0.0
|
|
self._starting_paused_off_since = None
|
|
self._delay_wait_high_start = None
|
|
# Item 501: a forced state is a fresh start, never a standby re-probe.
|
|
self._standby_reprobe = False
|
|
self._probe_hidden = False
|
|
self._probe_wait_since = None
|
|
self._probe_terminal_from = None
|
|
if carried is not None:
|
|
self._standby_reprobe, self._preroll_buffer = carried
|
|
|
|
@property
|
|
def state(self) -> str:
|
|
"""Return current state."""
|
|
return self._state
|
|
|
|
@property
|
|
def sub_state(self) -> str | None:
|
|
"""Return current sub-state."""
|
|
return self._sub_state
|
|
|
|
@property
|
|
def config(self) -> CycleDetectorConfig:
|
|
"""Return current configuration."""
|
|
return self._config
|
|
|
|
@property
|
|
def matched_profile(self) -> str | None:
|
|
"""Return the name of the matched profile, if any."""
|
|
return self._matched_profile
|
|
|
|
@property
|
|
def current_cycle_start(self) -> datetime | None:
|
|
"""Return the start timestamp of the current cycle."""
|
|
return self._current_cycle_start
|
|
|
|
@property
|
|
def samples_recorded(self) -> int:
|
|
"""Return the number of power samples recorded in current cycle."""
|
|
return len(self._power_readings)
|
|
|
|
@property
|
|
def expected_duration_seconds(self) -> float:
|
|
"""Return the expected duration of the current cycle in seconds."""
|
|
return self._expected_duration
|
|
|
|
@staticmethod
|
|
def _smart_term_block_reason(
|
|
current_duration: float,
|
|
expected: float,
|
|
smart_ratio: float,
|
|
is_confident: bool,
|
|
ambiguous: bool,
|
|
power_plausible: bool = True,
|
|
) -> str | None:
|
|
"""Why the Smart-Termination fast end-path did NOT fire, for diagnostics.
|
|
|
|
Returns None when the gate would pass, or when no expected duration is known
|
|
yet (nothing meaningful to report). Mirrors the gate's conditions in order so
|
|
the first blocking reason is surfaced. Pure and side-effect-free; the
|
|
detector logs the result (throttled to reason changes) - no behaviour
|
|
change (#346, extended with "still_active" for #364).
|
|
"""
|
|
if expected <= 0:
|
|
return None
|
|
if current_duration < expected * smart_ratio:
|
|
return "duration_not_reached"
|
|
if not is_confident:
|
|
return "low_confidence"
|
|
if ambiguous:
|
|
return "match_ambiguous"
|
|
if not power_plausible:
|
|
return "still_active"
|
|
return None
|
|
|
|
@staticmethod
|
|
def _resolve_smart_ratio(
|
|
device_type: str,
|
|
configured_ratio: float,
|
|
end_spike_seen: bool,
|
|
end_spike_duration: float,
|
|
expected_duration: float,
|
|
) -> float:
|
|
"""Resolve the Smart-Termination duration-ratio gate (#393).
|
|
|
|
``configured_ratio`` is the per-device option, already resolved in the
|
|
config builder to the device-type default (0.99 dishwasher / 0.98 other)
|
|
unless the user tuned it - so it is always a real float here.
|
|
|
|
For a dishwasher whose most-recent in-ENDING spike landed at >=90% of the
|
|
expected duration, that spike is the terminal pump-out (not a mid-cycle
|
|
rinse drain): once it is confirmed the gate is loosened to the 0.90
|
|
pump-out relief, because individual cycles can be a few % shorter than the
|
|
rolling average and still terminate cleanly. Keeping the configured gate
|
|
for spikes at <90% prevents premature closes during the passive Dry phase
|
|
that follows the pre-final-rinse drain. The relief is combined with the
|
|
configured value via ``min()`` so a configured ratio can only ever LOOSEN
|
|
the gate, never tighten it. Pure and side-effect-free (unit-testable).
|
|
"""
|
|
# Clamp to the documented [0.50, 1.00] range (the WS write path clamps, but a
|
|
# value persisted by an import or an older schema is read here unclamped): a
|
|
# 0.0 would drop the duration floor entirely and let Smart Termination fire the
|
|
# moment its other conditions pass.
|
|
configured_ratio = min(1.0, max(0.5, configured_ratio))
|
|
if (
|
|
device_type == "dishwasher"
|
|
and end_spike_seen
|
|
and expected_duration > 0
|
|
and end_spike_duration >= expected_duration * 0.90
|
|
):
|
|
return min(configured_ratio, 0.90)
|
|
return configured_ratio
|
|
|
|
def process_reading(
|
|
self,
|
|
power: float,
|
|
timestamp: datetime,
|
|
synthetic: bool = False,
|
|
observed: bool = True,
|
|
) -> None:
|
|
"""Process a new power reading using robust dt-aware logic.
|
|
|
|
``synthetic=True`` marks a reading the *manager* injected rather than one
|
|
the power sensor sent: the watchdog and anti-wrinkle keepalives, which
|
|
exist to advance the quiet timers while a change-only plug says nothing.
|
|
They must keep doing exactly that, so this flag does not change the quiet
|
|
accumulators. What it does change is the two places that reason about what
|
|
the SENSOR did (#424):
|
|
|
|
* ``_update_cadence`` is skipped. The cadence estimate feeds
|
|
``_gate_cadence`` and therefore the pause/end gates, so training it on
|
|
our own injections makes those gates a function of how often we inject -
|
|
see the note on ``_gate_cadence``.
|
|
* **with ``observed=True``**, the gap-free tally treats the interval as
|
|
observed, because the watchdog re-anchored on the sensor's live state
|
|
(``_resync_power_from_state``) before injecting - so the keepalive is a
|
|
moment we looked rather than a hole in the record.
|
|
|
|
``observed=False`` says the sensor state could NOT be read when this
|
|
keepalive was injected (unavailable / unknown / non-finite). The second
|
|
bullet is then false: the interval it closes is a genuine outage, and the
|
|
gap-free tally is reset like any other hole - at ANY step size, since the
|
|
watchdog injects far more often than the outage ceiling.
|
|
|
|
A RECORDED outage (``mark_sensor_unavailable``, register item 266) goes
|
|
further, because only it says where the hole began: from its start until
|
|
the next real reading, time is not added to ``_time_below_threshold``
|
|
either, so an outage can no longer run out an end gate. A keepalive inside
|
|
one advances the clock and nothing else - no state-machine step, no
|
|
sample in the trace - so no decision rests on a value nobody read. The
|
|
manager's staleness force-stop still ends a cycle whose sensor stays dead.
|
|
``observed=False`` alone (a late watchdog tick, item 391) keeps its
|
|
narrower meaning: it resets the gap-free tally only.
|
|
|
|
(The `_keep_tail_cap` use this flag was originally added for, register
|
|
item 238, was implemented, measured and reverted: after
|
|
``_last_active_time`` every reading is below the stop threshold anyway, so
|
|
a plug still reporting cannot separate a drying phase from standby - it
|
|
only shows the plug is chatty. See item 260 for two more that were
|
|
measured and rejected.)
|
|
"""
|
|
# Every interval below is a datetime subtraction. Two aware datetimes that
|
|
# share one tzinfo instance - which every `dt_util.now()` stamp does - are
|
|
# subtracted on their WALL-CLOCK fields, so across a DST change dt, elapsed
|
|
# and the stored duration are off by an hour (a spring-forward soak was
|
|
# credited 3960 s of quiet and split the wash; audit DETECT-01). UTC has no
|
|
# transitions, and mixed-zone comparisons elsewhere stay correct.
|
|
timestamp = dt_util.as_utc(timestamp)
|
|
outage_since = self._sensor_outage_since
|
|
if not synthetic:
|
|
self._last_real_reading_time = timestamp
|
|
if outage_since is not None:
|
|
# The sensor spoke: the outage is over (item 266).
|
|
self._sensor_outage_since = None
|
|
self._logger.debug(
|
|
"Power sensor reporting again after %.0fs unavailable; that "
|
|
"span was not counted as quiet",
|
|
(timestamp - outage_since).total_seconds(),
|
|
)
|
|
|
|
# Calculate dt (needed by the stop lockout below and the state machine).
|
|
dt = 0.0
|
|
if self._last_process_time:
|
|
dt = (timestamp - self._last_process_time).total_seconds()
|
|
|
|
# Sanity check for negative dt
|
|
if dt < 0:
|
|
# Logged because this is the one exit from process_reading that leaves
|
|
# NO trace: the accumulators do not advance, `_power_readings` does not
|
|
# grow, and every downstream gate therefore stays silent too. A cycle
|
|
# wedged behind it looks exactly like a cycle whose end gate is simply
|
|
# not satisfied - which is how much of register item 320 was spent.
|
|
self._logger.debug(
|
|
"Ignoring reading %.1fW: timestamp %s is %.1fs before the last "
|
|
"processed reading %s",
|
|
power,
|
|
timestamp,
|
|
-dt,
|
|
self._last_process_time,
|
|
)
|
|
self._last_process_time = timestamp
|
|
return
|
|
|
|
# Manual Stop Lockout:
|
|
# If user/external stop forced an end, ignore the machine's spin-down so
|
|
# it is not logged as a new cycle. The lockout clears the moment power
|
|
# drops to idle. As a safety net, if power instead stays high far longer
|
|
# than any plausible spin-down, treat it as a genuinely new back-to-back
|
|
# load and release the lockout so the cycle is detected immediately
|
|
# rather than pinned until the progress-reset window expires (#267).
|
|
if self._ignore_power_until_idle:
|
|
if power < self._config.start_threshold_w:
|
|
self._ignore_power_until_idle = False
|
|
self._lockout_high_seconds = 0.0
|
|
self._logger.debug(
|
|
"Power dropped below start threshold. Manual stop lockout cleared."
|
|
)
|
|
else:
|
|
self._lockout_high_seconds += dt
|
|
if self._lockout_high_seconds < STOP_LOCKOUT_RELEASE_SECONDS:
|
|
# Still within the spin-down window - ignore reading.
|
|
# The reading is withheld from the state machine, but it is
|
|
# still a real observation of the power level, so record it
|
|
# (#403): the accumulator below judges each interval against
|
|
# the previous observation, and a release reading compared to
|
|
# a pre-stop sample up to the full lockout window old would
|
|
# lose the credit for its own interval (#267 back-to-back
|
|
# start whose stop happened in a low-power trough).
|
|
self._last_process_time = timestamp
|
|
self._last_power = power
|
|
return
|
|
self._ignore_power_until_idle = False
|
|
self._lockout_high_seconds = 0.0
|
|
self._logger.info(
|
|
"Manual stop lockout released after sustained power "
|
|
"(>= %.1fs at/above start threshold): treating as a new "
|
|
"cycle (#267).",
|
|
STOP_LOCKOUT_RELEASE_SECONDS,
|
|
)
|
|
# Fall through: the state machine will start a new cycle.
|
|
|
|
# Snapshot the cadence BEFORE folding this reading in: the gap-free tally
|
|
# below classifies `dt` against a ceiling derived from the cadence, and an
|
|
# outage that has already widened p95 would raise the very threshold that
|
|
# is supposed to catch it (a 120 s gap after a 10 s cadence lifts p95 to
|
|
# ~15.5 s -> ceiling 155 s -> the gap counts as observed quiet).
|
|
self._prior_p95_dt = self._p95_dt
|
|
# Below five intervals `_p95_dt` is just the last interval (or the 1 s
|
|
# default), not a cadence, so nothing can be called outage-sized yet.
|
|
prior_cadence_known = len(self._recent_dts) >= 5
|
|
# Only the SENSOR trains the cadence estimator (#424). The watchdog's 0 W
|
|
# keepalives exist because the plug fell silent, so once it does every
|
|
# interval left in `_recent_dts` is one we manufactured: p95 AND the
|
|
# median both collapse onto the injection spacing, the `5 x median` cap in
|
|
# `_gate_cadence` stops binding, and the pause/end gates - three times that
|
|
# cadence - grow with our own injection rate. Measured on the #424
|
|
# reporter's v0.5.6 cycle: the gate climbed 192 s -> 530 s on injected
|
|
# readings alone, taking the end gate to 1605 s and holding PAUSED for
|
|
# 1060 s. Skipping them leaves the estimate describing the plug, which is
|
|
# the only thing it is supposed to describe; the accumulators below still
|
|
# advance on every reading, synthetic or not.
|
|
if not synthetic:
|
|
self._update_cadence(dt)
|
|
self._last_process_time = timestamp
|
|
|
|
# 1b. Pre-roll buffer (#430): record every reading seen while no cycle is
|
|
# open, so a start that needed several probes can recover what the aborted
|
|
# ones took with them. Cheap and bounded; inert while the option is off.
|
|
self._record_preroll(power, timestamp)
|
|
|
|
# 2. Accumulators Update
|
|
# Hysteresis Logic
|
|
if self._state in (STATE_OFF, STATE_DELAY_WAIT, STATE_STARTING, STATE_UNKNOWN):
|
|
threshold = self._config.start_threshold_w
|
|
elif self._state in (STATE_FINISHED, STATE_INTERRUPTED, STATE_FORCE_STOPPED):
|
|
# No cycle is open here either, so a terminal state starts on
|
|
# start_threshold_w like OFF (register item 510). It fell to the stop
|
|
# threshold by omission when these states were added: a display left
|
|
# on above stop probed STARTING on its first reading, a probe that
|
|
# cannot commit (STARTING measures against start_threshold_w), and the
|
|
# manager cleared Finished, Clean and the unload nag on it. The higher
|
|
# of the two, so an inverted pair (stop above start, seen in a
|
|
# contributed export) is not made easier to leave than before.
|
|
threshold = max(self._config.start_threshold_w, self._config.stop_threshold_w)
|
|
else:
|
|
threshold = self._config.stop_threshold_w
|
|
|
|
is_high = power >= threshold
|
|
|
|
# Last observation carried forward (#403): `dt` is the interval that
|
|
# ENDED at this reading, so the appliance sat at the PREVIOUS sample's
|
|
# level for it, not at this one. With a change-only (send-on-delta)
|
|
# power sensor a low -> high crossing carries the whole idle gap, and
|
|
# crediting it at the new high power let a single blip after minutes of
|
|
# silence satisfy both start gates on the next reading. So the interval
|
|
# only counts as high-power evidence when the previous observation was
|
|
# also at or above the threshold those gates measure against. A densely
|
|
# sampled device is unaffected: there the previous sample is already
|
|
# high and the interval keeps its full credit.
|
|
#
|
|
# This is the same principle the surrounding code already applies - the
|
|
# low branch restarts its gap-free tally rather than credit an outage,
|
|
# DELAY_WAIT and the paused-STARTING anchor (#306) anchor on the first
|
|
# high reading, and `integrate_wh`/`energy_gap_threshold_s` drop
|
|
# outage-sized segments - applied to the one branch that still credited
|
|
# unobserved time. An outage heuristic cannot substitute for it: a
|
|
# 511 s gap on a 70 s idle cadence is legitimate change-only silence,
|
|
# well inside the outage ceiling, and only the credit direction
|
|
# separates it from real high-power time.
|
|
#
|
|
# A reading inside the hysteresis band (>= stop_threshold_w but
|
|
# < start_threshold_w) therefore earns no evidence toward the start
|
|
# gates, which is correct: the band is by definition below the
|
|
# threshold the gates measure against, and it is exactly where a
|
|
# waiting machine idles. The cost is one extra report before
|
|
# confirmation on a band-crossing ramp; no start is lost.
|
|
prev_high = self._last_power is not None and self._last_power >= threshold
|
|
|
|
# The part of `dt` inside a recorded sensor outage (item 266, audit
|
|
# DETECT-13): from the outage start - or the previous reading, if later -
|
|
# up to this one. 0.0 whenever the sensor never went unavailable.
|
|
unobserved = 0.0
|
|
if outage_since is not None and dt > 0:
|
|
unobserved = min(dt, max(0.0, (timestamp - outage_since).total_seconds()))
|
|
# Outage-sized interval: clip(10x cadence, 60, 3600) like
|
|
# energy_gap_threshold_s, from the maintained p95 (O(1) in this hot path)
|
|
# as it stood BEFORE this reading, so a gap cannot widen its own ceiling.
|
|
outage_ceiling = min(3600.0, max(60.0, 10.0 * self._prior_p95_dt))
|
|
|
|
# Carrying the previous level forward assumes somebody was looking (item
|
|
# 266). Time inside a recorded outage, and a sensor interval the cadence
|
|
# calls an outage, is not high-power evidence either: crediting it let a
|
|
# plug that dropped out mid-heat confirm a start and bank its last power
|
|
# for the whole hole, which the stored energy (`integrate_wh` drops
|
|
# outage-sized segments) never counted. Synthetic keepalives are exempt
|
|
# from the size test for the reason given in the low branch below.
|
|
high_dt = 0.0
|
|
if prev_high:
|
|
high_dt = dt - unobserved
|
|
if not synthetic and prior_cadence_known and dt > outage_ceiling:
|
|
high_dt = 0.0
|
|
# ...and the ENERGY for that interval at the level the appliance actually sat
|
|
# at, which is the same argument applied to the second start gate. Crediting
|
|
# it at the NEW reading's power let a sample barely above the threshold,
|
|
# followed by a spike, bank the spike's power for the whole preceding
|
|
# interval and satisfy start_energy_threshold on its own. Computed here, not
|
|
# at the three use sites, because `self._last_power` is overwritten a few
|
|
# lines below - before the two STARTING seeds further down would read it.
|
|
# The sibling paths already do this: the DELAY_WAIT seed is this same
|
|
# accumulator over its high streak (item 504) and the anti-wrinkle window
|
|
# uses the trapezoid average.
|
|
high_step_wh = (
|
|
(self._last_power or 0.0) * (high_dt / 3600.0) if high_dt > 0 else 0.0
|
|
)
|
|
|
|
if is_high:
|
|
self._time_above_threshold += high_dt
|
|
self._time_below_threshold = 0.0
|
|
self._time_below_threshold_gapfree = 0.0
|
|
self._time_below_unobserved = 0.0
|
|
# Energy for the guarded interval, computed with high_dt above.
|
|
self._energy_since_idle_wh += high_step_wh
|
|
self._last_active_time = timestamp
|
|
else:
|
|
# Quiet nobody saw is not quiet (item 266). The watchdog keeps
|
|
# injecting while the sensor is unavailable, and crediting those
|
|
# ticks let a dropout that began on a low reading run out the
|
|
# fallback timeout: a washer was closed `completed` mid-wash and the
|
|
# rest recorded as a second cycle. Frozen, not reset, so the quiet
|
|
# observed on either side still counts and `is_waiting_low_power`
|
|
# keeps the manager's staleness force-stop armed for a dead sensor.
|
|
self._time_below_threshold += dt - unobserved
|
|
self._time_below_unobserved += unobserved
|
|
# Gap-free tally: an outage-sized step is unobserved time, so restart
|
|
# the observed-quiet tally from this sample instead of crediting the
|
|
# gap. Ceiling mirrors energy_gap_threshold_s (clip(10x cadence, 60,
|
|
# 3600)) but reuses the maintained p95 cadence to stay O(1) in this
|
|
# per-reading hot path. Uses the cadence as it stood BEFORE this
|
|
# reading, so a gap cannot widen its own acceptance threshold.
|
|
# A synthetic keepalive is normally not an outage: the watchdog
|
|
# resyncs against the sensor's live state before injecting, so the
|
|
# interval it closes IS observed. This matters because the ceiling is
|
|
# derived from the p95 cadence, which synthetic readings no longer
|
|
# train - without the exemption a 106 s keepalive on a 2 s-cadence
|
|
# plug would look like a 106 s hole and reset the tally on every
|
|
# tick, starving the two consumers that can only ever SHORTEN the
|
|
# wait (the dishwasher end-spike quiet release and the ENDING hard
|
|
# finalize).
|
|
#
|
|
# `observed` is what makes that premise true rather than assumed.
|
|
# `_resync_power_from_state` returns early when the sensor is
|
|
# unavailable / unknown / non-finite, but the watchdog injects anyway
|
|
# - it only checks the silence interval. So during a real telemetry
|
|
# outage every keepalive was exempt and the gap-free tally grew
|
|
# through quiet nobody ever saw, which is exactly what that tally
|
|
# exists not to count. The caller now says whether the sensor state
|
|
# could actually be read, and an unread sensor is an outage.
|
|
# An unread sensor is an outage HOWEVER SHORT each keepalive step
|
|
# is, which is why this is its own clause and not a qualifier on the
|
|
# ceiling test. The watchdog injects once per `watchdog_interval`
|
|
# (floor 30 s, effective 30-60 s) and the ceiling is at least 60 s,
|
|
# so during a real outage every individual `dt` sits under the
|
|
# ceiling - qualifying the ceiling test left the tally accumulating
|
|
# exactly as before, which is the bug this is meant to fix.
|
|
# (`outage_ceiling` is computed once, above the high-power credit.)
|
|
if (
|
|
(synthetic and not observed)
|
|
or unobserved > 0
|
|
or (dt > outage_ceiling and not synthetic)
|
|
):
|
|
self._time_below_threshold_gapfree = 0.0
|
|
else:
|
|
self._time_below_threshold_gapfree += dt
|
|
self._time_above_threshold = 0.0
|
|
|
|
self._time_in_state += dt
|
|
|
|
self._last_power = power
|
|
if power < self._config.stop_threshold_w:
|
|
# Back on the idle floor: the next probe is a fresh one (item 501).
|
|
self._standby_reprobe = False
|
|
if not synthetic:
|
|
# #452 idle display: real readings only (a keepalive is not a level).
|
|
self._update_idle(timestamp, power)
|
|
|
|
# A keepalive inside a recorded outage carries no observation, only the
|
|
# clock (item 266). Stepping the state machine on it would hand every
|
|
# finisher a window of the 0 W the manager injects for an unreadable
|
|
# sensor - the anti-crease finalize reads exactly such a window - and
|
|
# write that invented 0 W into the trace and the matcher's input. So it
|
|
# stops here; the next real reading picks up from the frozen tallies.
|
|
if (
|
|
synthetic
|
|
and outage_since is not None
|
|
and self._state in (STATE_STARTING, STATE_RUNNING, STATE_PAUSED, STATE_ENDING)
|
|
):
|
|
return
|
|
|
|
anti_wrinkle_active = (
|
|
self._config.anti_wrinkle_enabled
|
|
and self._config.device_type in (
|
|
DEVICE_TYPE_WASHING_MACHINE,
|
|
DEVICE_TYPE_DRYER,
|
|
DEVICE_TYPE_WASHER_DRYER,
|
|
)
|
|
)
|
|
|
|
# 3. State Machine
|
|
|
|
if self._state in (
|
|
STATE_OFF,
|
|
STATE_FINISHED,
|
|
STATE_INTERRUPTED,
|
|
STATE_FORCE_STOPPED,
|
|
STATE_ANTI_WRINKLE,
|
|
):
|
|
started_from_anti_wrinkle = False
|
|
tail_floor = (
|
|
self._anticrease_tail_floor_w
|
|
if self._state == STATE_ANTI_WRINKLE
|
|
else None
|
|
)
|
|
if (
|
|
anti_wrinkle_active
|
|
and tail_floor is not None
|
|
and is_high
|
|
and power <= tail_floor
|
|
):
|
|
# Register item 393a: the #296 tail's own baseline between bursts.
|
|
# It can sit above stop_threshold (the 3.3 W Knitterschutz draw
|
|
# against a ~1.5 W threshold), where it never reset the candidate,
|
|
# so every tail left anti-wrinkle after anti_wrinkle_max_duration.
|
|
self._anti_wrinkle_candidate_start = None
|
|
self._anti_wrinkle_candidate_peak = 0.0
|
|
self._anti_wrinkle_candidate_start_power = 0.0
|
|
elif anti_wrinkle_active and self._state == STATE_ANTI_WRINKLE and is_high:
|
|
if self._anti_wrinkle_candidate_start is None:
|
|
self._anti_wrinkle_candidate_start = timestamp
|
|
self._anti_wrinkle_candidate_peak = power
|
|
self._anti_wrinkle_candidate_start_power = power
|
|
else:
|
|
self._anti_wrinkle_candidate_peak = max(
|
|
self._anti_wrinkle_candidate_peak, power
|
|
)
|
|
|
|
candidate_duration = (
|
|
timestamp - self._anti_wrinkle_candidate_start
|
|
).total_seconds()
|
|
# Register item 393a: after the #296 finalise the tail is a train of
|
|
# sub-max-power drum bursts, and a send-on-change plug can skip the
|
|
# baseline between two of them (the reporter's 20 Knitterschutz
|
|
# tails all hold 60-111 s without a reading under stop_threshold),
|
|
# so the configured burst length opened a second cycle on the tail
|
|
# itself. The finalise accepted ANTI_CREASE_CONFIRM_WINDOW_S of such
|
|
# readings as tail; leaving it on duration takes a burst longer than
|
|
# that. A next wash's heating still leaves at once on the power test.
|
|
burst_limit = float(self._config.anti_wrinkle_max_duration)
|
|
if tail_floor is not None:
|
|
burst_limit = max(burst_limit, ANTI_CREASE_CONFIRM_WINDOW_S)
|
|
exceeds = (
|
|
self._anti_wrinkle_candidate_peak
|
|
> self._config.anti_wrinkle_max_power
|
|
or power > self._config.anti_wrinkle_max_power
|
|
or candidate_duration > burst_limit
|
|
)
|
|
|
|
if exceeds:
|
|
candidate_start = self._anti_wrinkle_candidate_start
|
|
candidate_peak = self._anti_wrinkle_candidate_peak
|
|
candidate_start_power = self._anti_wrinkle_candidate_start_power
|
|
self._anti_wrinkle_candidate_start = None
|
|
self._anti_wrinkle_candidate_peak = 0.0
|
|
self._anti_wrinkle_candidate_start_power = 0.0
|
|
self._transition_to(STATE_STARTING, timestamp)
|
|
started_from_anti_wrinkle = True
|
|
self._current_cycle_start = candidate_start or timestamp
|
|
|
|
# Preserve the anti-wrinkle candidate window instead of dropping ramp-up samples.
|
|
if candidate_start and candidate_start < timestamp:
|
|
start_power = candidate_start_power if candidate_start_power > 0 else power
|
|
self._power_readings = [(candidate_start, start_power), (timestamp, power)]
|
|
interval_s = (timestamp - candidate_start).total_seconds()
|
|
avg_power = (start_power + power) / 2.0
|
|
self._energy_since_idle_wh = max(0.0, avg_power * (interval_s / 3600.0))
|
|
else:
|
|
self._power_readings = [(timestamp, power)]
|
|
# Guarded interval (#403): the gap between anti-wrinkle
|
|
# tumbles was spent at the previous (idle) level.
|
|
self._energy_since_idle_wh = high_step_wh
|
|
|
|
self._cycle_max_power = max(candidate_peak, power)
|
|
elif self._state != STATE_ANTI_WRINKLE:
|
|
self._anti_wrinkle_candidate_start = None
|
|
self._anti_wrinkle_candidate_peak = 0.0
|
|
self._anti_wrinkle_candidate_start_power = 0.0
|
|
|
|
if self._state == STATE_ANTI_WRINKLE:
|
|
# Track time in idle. Quiet is below the exit power OR the stop
|
|
# threshold, whichever is higher, so the exit power only matters when
|
|
# set above stop. Not a "true off" level below stop (#285 #296 #325):
|
|
# after 33 of 139 corpus washer/dryer cycles the appliance idled
|
|
# between the 0.8 W default and its stop threshold, which would hold
|
|
# anti-wrinkle (and with it Clean and the unload reminder) for the
|
|
# 2 h cap. The #296 tail's baseline above stop is the tail floor's job.
|
|
effective_exit = max(self._config.anti_wrinkle_exit_power, self._config.stop_threshold_w)
|
|
if power < effective_exit:
|
|
# Low-power gap invalidates any burst candidate collected while in anti-wrinkle.
|
|
self._anti_wrinkle_candidate_start = None
|
|
self._anti_wrinkle_candidate_peak = 0.0
|
|
self._anti_wrinkle_candidate_start_power = 0.0
|
|
self._anti_wrinkle_idle_time += dt
|
|
anti_wrinkle_end_threshold = max(
|
|
self._dynamic_end_threshold,
|
|
float(self._config.anti_wrinkle_idle_timeout),
|
|
)
|
|
if self._anti_wrinkle_idle_time >= anti_wrinkle_end_threshold:
|
|
self._transition_to(STATE_OFF, timestamp)
|
|
return
|
|
else:
|
|
# Reset idle timer when power rises (burst detected)
|
|
self._anti_wrinkle_idle_time = 0.0
|
|
|
|
# Exit conditions:
|
|
# 1. Idle duration exceeded (handled above), OR
|
|
# 2. Safety timeout (2 hours in anti-wrinkle), OR
|
|
# 3. External trigger (user_stop, external triggers handled by manager)
|
|
if (
|
|
self._state_enter_time
|
|
and (timestamp - self._state_enter_time).total_seconds() > 7200
|
|
):
|
|
# Safety timeout: 2 hours in anti-wrinkle
|
|
self._transition_to(STATE_OFF, timestamp)
|
|
return
|
|
|
|
# Delayed-start "standby band" detection (only from STATE_OFF).
|
|
#
|
|
# A machine in delayed-start mode sits in a power band between
|
|
# the off-noise floor (stop_threshold_w) and the cycle-start
|
|
# threshold (start_threshold_w) - display, electronics, the
|
|
# occasional anti-damp tumble - for minutes to hours. We
|
|
# track anchored elapsed time while power is in that band; once
|
|
# it crosses delay_confirm_seconds we transition to DELAY_WAIT.
|
|
#
|
|
# Brief high-power excursions (menu navigation, button presses)
|
|
# don't break the candidate: they fall through to the normal
|
|
# start logic below, and unless they sustain for
|
|
# start_duration_threshold they get aborted as a false start
|
|
# and we re-enter the band on the next reading. Excursions
|
|
# below stop_threshold_w (machine momentarily idle on the noise
|
|
# floor) DO reset the candidate, because that's the same
|
|
# signal we use to define "off".
|
|
if (
|
|
self._config.delay_detect_enabled
|
|
and self._state == STATE_OFF
|
|
and not started_from_anti_wrinkle
|
|
and self._config.stop_threshold_w < self._config.start_threshold_w
|
|
):
|
|
in_band = (
|
|
self._config.stop_threshold_w
|
|
<= power
|
|
< self._config.start_threshold_w
|
|
)
|
|
if in_band:
|
|
if self._delay_band_start is None:
|
|
self._delay_band_start = timestamp
|
|
self._delay_band_seconds = 0.0
|
|
else:
|
|
self._delay_band_seconds = (
|
|
timestamp - self._delay_band_start
|
|
).total_seconds()
|
|
self._delay_band_peak = max(self._delay_band_peak, power)
|
|
if self._delay_band_seconds >= self._config.delay_confirm_seconds:
|
|
self._logger.info(
|
|
"Delayed start detected: standby band held for %.0fs "
|
|
"(peak %.1fW, current %.1fW) → DELAY_WAIT",
|
|
self._delay_band_seconds,
|
|
self._delay_band_peak,
|
|
power,
|
|
)
|
|
self._transition_to(STATE_DELAY_WAIT, timestamp)
|
|
return
|
|
# Stay in OFF while we accumulate evidence - do not
|
|
# fall through to the high-power start logic, the
|
|
# reading is below threshold by definition.
|
|
return
|
|
elif power < self._config.stop_threshold_w:
|
|
# Machine genuinely idle: forget any band history.
|
|
self._delay_band_start = None
|
|
self._delay_band_seconds = 0.0
|
|
self._delay_band_peak = 0.0
|
|
self._preserve_delay_band_on_off = False
|
|
# power >= start_threshold_w: fall through to the normal
|
|
# start path below. If it turns out to be a brief peak,
|
|
# STATE_STARTING will abort it as a false start and we'll
|
|
# re-enter the band check on the next sample without
|
|
# losing accumulated time (we don't reset on a high
|
|
# excursion - most users' "menu navigation" peaks last
|
|
# less than a sample interval anyway).
|
|
|
|
terminal = self._state in (
|
|
STATE_FINISHED, STATE_INTERRUPTED, STATE_FORCE_STOPPED
|
|
)
|
|
if terminal and not is_high and power >= self._config.stop_threshold_w:
|
|
# Item 510: a terminal state holds through the standby band instead
|
|
# of probing on it, so the band reading marks the next probe as a
|
|
# standby re-probe (item 501), as the band probe's abort used to.
|
|
self._standby_reprobe = True
|
|
|
|
if is_high and not started_from_anti_wrinkle:
|
|
# Transition to STARTING
|
|
self._preserve_delay_band_on_off = self._delay_band_start is not None
|
|
self._hide_probe(
|
|
self._standby_reprobe and (self._state == STATE_OFF or terminal)
|
|
)
|
|
back = (
|
|
(self._state, self._state_enter_time, self._sub_state)
|
|
if terminal and TERMINAL_PROBE_RETURNS
|
|
else None
|
|
)
|
|
self._transition_to(STATE_STARTING, timestamp)
|
|
self._probe_terminal_from = back # item 515
|
|
self._current_cycle_start = timestamp
|
|
self._power_readings = [(timestamp, power)]
|
|
# Seed from the guarded interval (#403), not raw dt: this seed
|
|
# OVERWRITES the accumulator (it has to - entering STARTING from
|
|
# a terminal state carries the previous cycle's total), so an
|
|
# unguarded seed would reinstate the idle gap the accumulator
|
|
# just declined to credit.
|
|
self._energy_since_idle_wh = high_step_wh
|
|
self._cycle_max_power = power
|
|
self._apply_curve_preroll(timestamp, power)
|
|
self._show_probe_with_evidence()
|
|
# NOTE: terminal-state expiry (Finished/Interrupted/Force-Stopped -> Off)
|
|
# is owned solely by the manager (WashDataManager._handle_state_expiry),
|
|
# which has a wall-clock timer that also fires when a change-only power
|
|
# sensor stops reporting, plus the opt-in power-based Off (issue #284).
|
|
# The detector used to auto-expire here after a hardcoded 30 min, but that
|
|
# duplicated the manager timer (a weaker, per-reading subset) and left the
|
|
# manager's bookkeeping (progress, clean overlay, notifications) dangling.
|
|
# ANTI_WRINKLE -> Off is handled by its own idle/timeout logic above.
|
|
|
|
elif self._state == STATE_DELAY_WAIT:
|
|
if power >= self._config.start_threshold_w:
|
|
# Power is in cycle-start territory. Require at least
|
|
# two consecutive high readings spanning
|
|
# start_duration_threshold real seconds before committing
|
|
# to STARTING, so a single isolated spike (a heavy menu
|
|
# interaction, an anti-damp pulse briefly crossing the
|
|
# threshold) doesn't false-trigger. We anchor on the
|
|
# FIRST high reading instead of accumulating dt, because
|
|
# dt to the previous (low) reading is unrelated to how
|
|
# long the high power has actually persisted.
|
|
self._delay_wait_true_off_seconds = 0.0
|
|
if self._delay_wait_high_start is None:
|
|
self._delay_wait_high_start = timestamp
|
|
self._delay_wait_high_power = power
|
|
# Item 504: the streak's energy from here, credited per guarded
|
|
# interval by the accumulator above (#403), seeds the probe.
|
|
# The previous reading was not high, so nothing before the
|
|
# anchor is lost.
|
|
self._energy_since_idle_wh = 0.0
|
|
else:
|
|
elapsed_high = (
|
|
timestamp - self._delay_wait_high_start
|
|
).total_seconds()
|
|
if elapsed_high >= self._config.start_duration_threshold:
|
|
# Debug, like any other probe: since item 504 a straddling
|
|
# standby probes from here hundreds of times a day.
|
|
self._logger.debug(
|
|
"Delayed start: probing (power %.1fW sustained ≥ %.1fW for %.0fs)",
|
|
power,
|
|
self._config.start_threshold_w,
|
|
elapsed_high,
|
|
)
|
|
self._hide_probe(True) # out of a standby (item 501)
|
|
wait_since = self._state_enter_time or timestamp
|
|
self._transition_to(STATE_STARTING, timestamp)
|
|
self._probe_wait_since = wait_since # item 504
|
|
start_timestamp = self._delay_wait_high_start or timestamp
|
|
start_power = self._delay_wait_high_power or power
|
|
self._current_cycle_start = start_timestamp
|
|
self._power_readings = [(start_timestamp, start_power)]
|
|
# `_energy_since_idle_wh` already holds the streak's energy
|
|
# (item 504). It was `start_power * elapsed`, which credited
|
|
# the first high reading's level for the whole window: a
|
|
# 22 W blip then 5 W for 20 s banked 0.15 of a 0.2 Wh gate,
|
|
# and once false starts return here every standby probe
|
|
# started that way (max aborted probe 0.53 -> 0.78 of the
|
|
# gate on #35).
|
|
if timestamp != start_timestamp:
|
|
self._power_readings.append((timestamp, power))
|
|
self._cycle_max_power = max(start_power, power)
|
|
# #430: the buffer records in DELAY_WAIT too, so the
|
|
# readings between the anchor and this confirmation exist
|
|
# and would otherwise be dropped - the curve would span the
|
|
# confirmation window with two points. Never moves the
|
|
# anchor forward (see _apply_curve_preroll), and is a no-op
|
|
# while the option is off, which is the default.
|
|
self._apply_curve_preroll(timestamp, power)
|
|
self._show_probe_with_evidence()
|
|
else:
|
|
# Power dropped back below start threshold - clear the
|
|
# high-power streak anchor so the next high reading
|
|
# starts a fresh confirmation window.
|
|
self._delay_wait_high_start = None
|
|
self._delay_wait_high_power = None
|
|
if power < self._config.stop_threshold_w:
|
|
# Power near zero: machine genuinely turned off, not
|
|
# just waiting.
|
|
self._delay_wait_true_off_seconds += dt
|
|
if self._delay_wait_true_off_seconds >= 30.0:
|
|
self._logger.info(
|
|
"Delayed start cancelled: power dropped to off (%.1fW) for %.0fs",
|
|
power,
|
|
self._delay_wait_true_off_seconds,
|
|
)
|
|
self._transition_to(STATE_OFF, timestamp)
|
|
return
|
|
else:
|
|
self._delay_wait_true_off_seconds = 0.0
|
|
|
|
# Safety timeout
|
|
if (
|
|
self._state_enter_time
|
|
and (timestamp - self._state_enter_time).total_seconds()
|
|
>= self._config.delay_timeout_seconds
|
|
):
|
|
self._logger.info(
|
|
"Delayed start timeout after %.0fh → OFF",
|
|
self._config.delay_timeout_seconds / 3600.0,
|
|
)
|
|
self._transition_to(STATE_OFF, timestamp)
|
|
|
|
elif self._state == STATE_STARTING:
|
|
self._power_readings.append((timestamp, power))
|
|
self._cycle_max_power = max(self._cycle_max_power, power)
|
|
|
|
if is_high:
|
|
# Power back up - clear any accumulated "true off" hold time.
|
|
self._starting_paused_off_since = None
|
|
|
|
self._show_probe_with_evidence() # item 501
|
|
|
|
if self._time_above_threshold >= self._config.start_duration_threshold:
|
|
if self._energy_since_idle_wh >= self._config.start_energy_threshold:
|
|
self._transition_to(STATE_RUNNING, timestamp)
|
|
|
|
# Abort if power drops below threshold before confirmation.
|
|
# Skip the abort when the user has explicitly paused the cycle
|
|
# (issue #306): a user pause sets verified_pause=True, which signals
|
|
# that the low power is intentional, not a false start.
|
|
if not is_high and self._time_below_threshold > 1.0: # 1s grace period
|
|
if getattr(self, "_verified_pause", False):
|
|
# User pause holds; wait for Resume Cycle (issue #306). But a
|
|
# genuinely paused appliance keeps standby power above the stop
|
|
# threshold - sustained power *below* it means the machine was
|
|
# switched off, so fall back to OFF rather than pinning STARTING
|
|
# forever.
|
|
if power < self._config.stop_threshold_w:
|
|
# An outage-sized gap since the last reading is NOT observed
|
|
# quiet (the machine may still be paused, we just lost
|
|
# telemetry): reset anchor so only genuinely-sampled
|
|
# sustained-off time can cancel a paused STARTING.
|
|
if dt > self._outage_threshold_s():
|
|
self._starting_paused_off_since = None
|
|
elif self._starting_paused_off_since is None:
|
|
# First below-stop reading: anchor the timestamp.
|
|
# Don't credit the preceding dt interval — we only
|
|
# know the device is off *now*, not how long before
|
|
# this sample it went quiet.
|
|
self._starting_paused_off_since = timestamp
|
|
observed_off_s = (
|
|
(timestamp - self._starting_paused_off_since).total_seconds()
|
|
if self._starting_paused_off_since is not None
|
|
else 0.0
|
|
)
|
|
if observed_off_s >= STARTING_PAUSED_TRUE_OFF_TIMEOUT_SECONDS:
|
|
self._logger.info(
|
|
"Paused STARTING cancelled: power off (%.1fW) for "
|
|
"%.0fs → OFF",
|
|
power,
|
|
observed_off_s,
|
|
)
|
|
self._transition_to(STATE_OFF, timestamp)
|
|
return
|
|
else:
|
|
# Power recovered — clear the off anchor.
|
|
self._starting_paused_off_since = None
|
|
else:
|
|
# False start
|
|
self._logger.debug(
|
|
"False start detected: power dropped after %.2fs",
|
|
self._time_above_threshold,
|
|
)
|
|
# Do NOT reset _delay_band_* here — _transition_to(STATE_OFF) will
|
|
# preserve the band via _preserve_delay_band_on_off if it was set
|
|
# at STARTING entry (line 838), so a brief high-power peak (menu
|
|
# navigation) doesn't restart the delayed-start accumulation from zero.
|
|
# Item 501: back in the band, not on the idle floor, so the next
|
|
# probe is a standby re-probe and is not shown until it has evidence.
|
|
self._standby_reprobe = power >= self._config.stop_threshold_w
|
|
wait_since = self._probe_wait_since
|
|
waited_s = (
|
|
(timestamp - wait_since).total_seconds()
|
|
if wait_since is not None
|
|
else 0.0
|
|
)
|
|
if self._probe_terminal_from is not None:
|
|
# Item 515: out of Finished / Interrupted / Force-Stopped, back
|
|
# there whatever the power: a probe that never committed ends
|
|
# nothing, and the manager's expiry still owns the way to OFF.
|
|
self._return_probe_to_terminal(timestamp, power)
|
|
elif (
|
|
wait_since is not None
|
|
and self._standby_reprobe
|
|
and waited_s < self._config.delay_timeout_seconds
|
|
):
|
|
# Item 504: a probe out of DELAY_WAIT that falls back into the
|
|
# band goes back to waiting, keeping DELAY_WAIT's timeout
|
|
# anchor. Back to OFF lost it, and on a standby straddling
|
|
# start_threshold_w the band (OFF only) re-armed DELAY_WAIT a
|
|
# minute later: off <-> waiting ~16 times per idle day on
|
|
# #35. A drop below stop_threshold_w still ends the wait,
|
|
# and so does the timeout, checked here because on such a
|
|
# standby this abort may be the only in-band reading.
|
|
self._transition_to(STATE_DELAY_WAIT, timestamp)
|
|
self._state_enter_time = wait_since
|
|
self._time_in_state = max(0.0, waited_s)
|
|
else:
|
|
if wait_since is not None and self._standby_reprobe:
|
|
self._logger.info(
|
|
"Delayed start timeout after %.0fh → OFF",
|
|
self._config.delay_timeout_seconds / 3600.0,
|
|
)
|
|
self._transition_to(STATE_OFF, timestamp)
|
|
|
|
elif self._state == STATE_RUNNING:
|
|
self._power_readings.append((timestamp, power))
|
|
self._cycle_max_power = max(self._cycle_max_power, power)
|
|
self._update_stall(timestamp, power) # #452, display + standby-band hold
|
|
|
|
# Anti-crease finalize (#296): a matched cycle past its expected
|
|
# duration that has settled into the low-power tumble tail is done -
|
|
# finalize into anti-wrinkle now instead of letting the periodic
|
|
# bursts keep reviving RUNNING until a second wash merges in.
|
|
if self._maybe_finalize_anticrease_tail(timestamp):
|
|
return
|
|
|
|
# Use dynamic threshold
|
|
thresh = self._dynamic_pause_threshold
|
|
if self._time_below_threshold >= thresh:
|
|
self._try_profile_match(timestamp, force=True) # Refine match on pause
|
|
self._transition_to(STATE_PAUSED, timestamp)
|
|
|
|
# Periodic profile matching
|
|
self._try_profile_match(timestamp)
|
|
|
|
# Standby-band stuck finalize (#296): an appliance holding a flat
|
|
# low standby draw ABOVE stop_threshold never accumulates
|
|
# _time_below_threshold, so it never reaches PAUSED/ENDING. Detect
|
|
# the plateau and finalize as a normal completion (so anti-wrinkle
|
|
# still engages). Cheaply gated on being well past expected before
|
|
# the window scan runs.
|
|
if self._maybe_finalize_standby_band(timestamp, power):
|
|
return
|
|
|
|
# Max duration safety
|
|
if (
|
|
self._current_cycle_start
|
|
and (timestamp - self._current_cycle_start).total_seconds() > 28800
|
|
): # 8h safety
|
|
self._finish_cycle(
|
|
timestamp,
|
|
status="force_stopped",
|
|
termination_reason=TerminationReason.FORCE_STOPPED,
|
|
)
|
|
|
|
elif self._state == STATE_PAUSED:
|
|
self._power_readings.append((timestamp, power))
|
|
self._update_stall(timestamp, power) # #452
|
|
|
|
# Anti-crease finalize (#296) - see the RUNNING branch.
|
|
if self._maybe_finalize_anticrease_tail(timestamp):
|
|
return
|
|
|
|
if is_high:
|
|
# Resume to RUNNING
|
|
self._transition_to(STATE_RUNNING, timestamp)
|
|
else:
|
|
# Periodic profile matching during pause
|
|
self._try_profile_match(timestamp)
|
|
|
|
thresh = self._dynamic_end_threshold
|
|
if self._time_below_threshold >= thresh:
|
|
self._transition_to(STATE_ENDING, timestamp)
|
|
|
|
elif self._state == STATE_ENDING:
|
|
self._power_readings.append((timestamp, power))
|
|
self._update_stall(timestamp, power) # #452
|
|
|
|
# Hard cap: ENDING must not run longer than RUNNING's 8 h safety limit.
|
|
# Without this a standby baseline can hold the state open indefinitely.
|
|
if (
|
|
self._current_cycle_start
|
|
and (timestamp - self._current_cycle_start).total_seconds() > 28800
|
|
):
|
|
self._finish_cycle(
|
|
timestamp,
|
|
status="force_stopped",
|
|
termination_reason=TerminationReason.FORCE_STOPPED,
|
|
)
|
|
return
|
|
|
|
# Anti-crease finalize (#296) - see the RUNNING branch. Fires ahead of
|
|
# the is_high end-spike handling so a sub-max_power tail burst finalizes
|
|
# into anti-wrinkle instead of reviving RUNNING.
|
|
if self._maybe_finalize_anticrease_tail(timestamp):
|
|
return
|
|
|
|
if is_high:
|
|
# Standby-band finalize (#296/#445) from ENDING too (DETECT-03). A
|
|
# run that dipped long enough to reach ENDING and THEN settled on a
|
|
# flat standby above stop_threshold keeps every reading "high":
|
|
# after 120 s in state each one is kept below as a terminal spike
|
|
# and returns, resetting the quiet timer, so neither the fallback
|
|
# nor the RUNNING-only plateau check could ever end it - the 8 h
|
|
# force-stop did. Same predicate and trim as RUNNING.
|
|
if self._maybe_finalize_standby_band(timestamp, power):
|
|
return
|
|
|
|
# Programme time: a stall is not progress (item 511).
|
|
current_duration = self._gate_elapsed_s(timestamp)
|
|
|
|
is_dishwasher = self._config.device_type == "dishwasher"
|
|
|
|
# Issue #43: only treat this as a *terminal* end spike (which then
|
|
# pre-arms Smart Termination) when it occurs near the end of the
|
|
# expected cycle. Mid-cycle spikes - e.g. the dishwasher
|
|
# wash→drying drain wind-down at ~50% of expected duration - must
|
|
# not arm smart termination, otherwise the cycle finishes at 99%
|
|
# of expected *before* the real end-of-cycle pump-out, and that
|
|
# pump-out is then misread as a brand-new cycle. Without a
|
|
# matched profile (expected==0) the gating is bypassed so the
|
|
# legacy "any spike counts" behaviour is preserved for unmatched
|
|
# cycles (relied on by the dishwasher unmatched-cap path).
|
|
if (
|
|
self._expected_duration <= 0
|
|
or current_duration
|
|
>= self._expected_duration * DISHWASHER_END_SPIKE_MIN_PROGRESS
|
|
) and self._spike_follows_terminal_quiet(timestamp):
|
|
self._end_spike_seen = True
|
|
self._end_spike_duration = current_duration
|
|
self._logger.debug(
|
|
"End spike detected (power high in ENDING state, "
|
|
"%.0fs/%.0fs)",
|
|
current_duration,
|
|
self._expected_duration,
|
|
)
|
|
else:
|
|
self._logger.debug(
|
|
"Mid-cycle spike in ENDING ignored for end-spike "
|
|
"tracking (%.0fs < %.0f%% of expected %.0fs)",
|
|
current_duration,
|
|
DISHWASHER_END_SPIKE_MIN_PROGRESS * 100,
|
|
self._expected_duration,
|
|
)
|
|
|
|
# Sanity check: if expected_duration is unreasonable (>6 hours), use fallback
|
|
max_reasonable = 21600.0 # 6 hours
|
|
effective_expected = self._expected_duration
|
|
|
|
if effective_expected <= 0 or effective_expected > max_reasonable:
|
|
# Fallback: use current duration + buffer if we've run > 3 hours
|
|
# (Assumes any cycle over 3 hours running is near completion when in ENDING)
|
|
if current_duration > 10800: # 3 hours
|
|
effective_expected = current_duration * 0.99 # Always past threshold
|
|
self._logger.debug(
|
|
"End spike check using fallback: expected_duration=%ds is unreasonable, "
|
|
"using current_duration=%ds as reference",
|
|
int(self._expected_duration), int(current_duration)
|
|
)
|
|
|
|
past_expected = (
|
|
effective_expected > 0
|
|
and current_duration >= (effective_expected * 0.98)
|
|
)
|
|
|
|
# If ENDING has already lasted long enough, treat any power burst as
|
|
# terminal (applies to all device types). Dishwashers additionally check
|
|
# proximity to the expected duration.
|
|
long_ending_tail = self._time_in_state >= 120.0
|
|
terminal_spike = long_ending_tail
|
|
|
|
if is_dishwasher:
|
|
near_expected = (
|
|
effective_expected > 0
|
|
and current_duration >= (effective_expected * 0.90)
|
|
)
|
|
terminal_spike = near_expected or long_ending_tail
|
|
|
|
if terminal_spike:
|
|
self._logger.debug(
|
|
"End spike kept in ENDING (duration %.0fs/%.0fs, time_in_ending %.0fs)",
|
|
current_duration,
|
|
effective_expected,
|
|
self._time_in_state,
|
|
)
|
|
return
|
|
|
|
if past_expected:
|
|
self._logger.debug(
|
|
"End spike ignored for state transition (past expected duration %.0fs/%.0fs)",
|
|
current_duration, effective_expected
|
|
)
|
|
# Stay in ENDING, the spike is recorded but doesn't resume cycle
|
|
else:
|
|
# Resume -> RUNNING (spike is genuine mid-cycle activity)
|
|
self._transition_to(STATE_RUNNING, timestamp)
|
|
else:
|
|
# Periodic profile matching during ending
|
|
self._try_profile_match(timestamp)
|
|
# A verified pause a revoke orphaned is released after a bounded
|
|
# quiet, so the fallback below can end the cycle (item 498).
|
|
self._release_orphaned_pause()
|
|
|
|
# --- SMART TERMINATION CHECK ---
|
|
# If we have a confident profile match and duration meets expectations,
|
|
# we terminate early (after appropriate debounce), ignoring long arbitrary timeouts.
|
|
if self._matched_profile:
|
|
start_time = self._current_cycle_start or timestamp
|
|
# Programme time (item 511): the halt itself carried the clock
|
|
# past the ratio, so the resumed wash ended at its next quiet.
|
|
current_duration = self._gate_elapsed_s(timestamp)
|
|
|
|
# --- ROBUSTNESS UPGRADE ---
|
|
# 1. Require higher duration ratio for Smart path
|
|
# 2. Require debounce to be measured FROM entry into ENDING state
|
|
|
|
# Per-appliance configurable gate (#393): the ratio is resolved
|
|
# in the config builder to the device-type default (0.99
|
|
# dishwasher / 0.98 other) unless the user tuned it, and the
|
|
# dishwasher pump-out relief is folded in via a pure helper so
|
|
# the gate logic stays unit-testable (see _resolve_smart_ratio).
|
|
smart_ratio = self._resolve_smart_ratio(
|
|
self._config.device_type,
|
|
self._config.smart_termination_duration_ratio,
|
|
getattr(self, "_end_spike_seen", False),
|
|
getattr(self, "_end_spike_duration", 0.0),
|
|
self._expected_duration,
|
|
)
|
|
|
|
is_confident_match = (
|
|
getattr(self, "_last_match_confidence", 0.0)
|
|
>= self._config.match_confidence_threshold
|
|
)
|
|
# Compute the #364 power-plausibility once per reading and reuse it
|
|
# for the diagnostic reason and the gate below - the helper walks the
|
|
# trailing window, so calling it two/three times per reading is waste.
|
|
_power_plausible = self._smart_term_power_plausible(timestamp)
|
|
|
|
# Gate the predictive end on match certainty.
|
|
# _match_ambiguous: top-1 vs top-2 score gap is too small to
|
|
# trust the matched profile's expected duration - fall through
|
|
# to the power-based fallback timeout instead. (The #364
|
|
# prefix-fit flag that also blocked here - "a longer candidate
|
|
# explains this trace better" - was removed in 0.5.8: since
|
|
# #400 the live matcher scores that prefix itself, and the flag
|
|
# never fired at a split moment on the corpus. See const.py.)
|
|
# Surface why the fast end-path is (not) firing, throttled to
|
|
# reason changes so a stuck cycle's cause is visible in the log
|
|
# without spamming every reading. Pure diagnostic (#346).
|
|
_block_reason = self._smart_term_block_reason(
|
|
current_duration,
|
|
self._expected_duration,
|
|
smart_ratio,
|
|
is_confident_match,
|
|
self._match_ambiguous,
|
|
_power_plausible,
|
|
)
|
|
if _block_reason != self._last_smart_term_block_reason:
|
|
self._last_smart_term_block_reason = _block_reason
|
|
if _block_reason is not None:
|
|
self._logger.debug(
|
|
"Smart Termination not applied (%s): dur=%.0fs/%.0fs conf=%.2f "
|
|
"ambiguous=%s trailing_power=%s profile_tail=%s",
|
|
_block_reason,
|
|
current_duration,
|
|
self._expected_duration * smart_ratio,
|
|
getattr(self, "_last_match_confidence", 0.0),
|
|
self._match_ambiguous,
|
|
self._trailing_mean_power(timestamp, self._tail_window_s()),
|
|
self._matched_tail_power,
|
|
)
|
|
|
|
if (
|
|
current_duration >= (self._expected_duration * smart_ratio)
|
|
and is_confident_match
|
|
and not self._match_ambiguous
|
|
# A user pause is authoritative: every other finisher honours
|
|
# it, and the paused time itself pushes elapsed past the
|
|
# ratio, so a washer paused near its end was finished by
|
|
# Smart Termination while the plug sat at 0 W (DETECT-05).
|
|
and not getattr(self, "_user_paused", False)
|
|
# #364: the clock says "done", but if we are still drawing
|
|
# several times what this profile draws at its own end, the
|
|
# match is a shorter look-alike and we are mid-wash. Block;
|
|
# the power-based fallback timeout decides instead.
|
|
and _power_plausible
|
|
):
|
|
# Dynamic confirmation window
|
|
if self._config.device_type == "dishwasher":
|
|
# Fixed - NOT off_delay-derived. off_delay is sized to
|
|
# bridge the long drying "pause", but must not delay the
|
|
# end; see DISHWASHER_SMART_TERMINATION_DEBOUNCE_SECONDS.
|
|
smart_debounce = DISHWASHER_SMART_TERMINATION_DEBOUNCE_SECONDS
|
|
elif self._config.device_type in (
|
|
DEVICE_TYPE_WASHING_MACHINE,
|
|
DEVICE_TYPE_WASHER_DRYER,
|
|
):
|
|
# Washing machines and washer-dryers have soak and
|
|
# rinse gaps that can dip for several minutes between
|
|
# programme phases. Require quiet time equal to half
|
|
# the soak-bridging min_off_gap before committing
|
|
# Smart Termination, so a near-duplicate profile
|
|
# doesn't cut a long cycle short during a mid-cycle
|
|
# power trough. Bounded above so a large suggested /
|
|
# hand-set min_off_gap can't inflate the quiet-time
|
|
# requirement and starve end-detection (see
|
|
# WASHER_SMART_TERMINATION_DEBOUNCE_MAX_SECONDS).
|
|
smart_debounce = min(
|
|
WASHER_SMART_TERMINATION_DEBOUNCE_MAX_SECONDS,
|
|
max(180.0, self._config.min_off_gap * 0.5),
|
|
)
|
|
else:
|
|
smart_debounce = 120.0
|
|
|
|
# Quiet time, not time in ENDING, for everything but a
|
|
# dishwasher: a wash that resumes after > 120 s in ENDING
|
|
# stays in ENDING, so `_time_in_state` had the debounce
|
|
# pre-paid by washing time and Smart fired on the next dip,
|
|
# turning the final spin into a second cycle (DETECT-04;
|
|
# 0/295 corpus cycles move). A dishwasher keeps state time,
|
|
# or every pump-out would add up to 300 s.
|
|
if self._config.device_type == "dishwasher":
|
|
_smart_quiet = self._time_in_state
|
|
else:
|
|
_smart_quiet = min(
|
|
self._time_in_state, self._time_below_threshold
|
|
)
|
|
if _smart_quiet >= smart_debounce:
|
|
# --- END SPIKE WAIT PERIOD (Dishwashers) ---
|
|
# Dishwashers should see the real end-of-cycle
|
|
# pump-out (which arms _end_spike_seen via the 85%
|
|
# progress gate) before Smart Termination fires -
|
|
# otherwise the pump-out arrives AFTER the cycle
|
|
# has already closed and registers as a brand-new
|
|
# "ghost" cycle. User reports (issue #43) showed
|
|
# the original 5-min past_wait_period escape hatch
|
|
# closing the cycle ~4 min before the real pump-out
|
|
# at ~99.5% of expected. Widen the escape hatch
|
|
# substantially (DISHWASHER_END_SPIKE_WAIT_SECONDS,
|
|
# currently 30 min past expected) so it cannot
|
|
# short-circuit a pump-out that fires within a
|
|
# reasonable window around expected end, but still
|
|
# guarantees the cycle terminates eventually for
|
|
# dishwashers that have no pump-out at all.
|
|
end_spike_seen = getattr(self, "_end_spike_seen", False)
|
|
# Release the pump-out wait once EITHER the cycle has run
|
|
# DISHWASHER_END_SPIKE_WAIT_SECONDS past its expected
|
|
# duration OR it has already reached its expected duration
|
|
# AND power has since stayed sustained-quiet for
|
|
# DISHWASHER_END_SPIKE_QUIET_RELEASE_SECONDS. The second arm
|
|
# closes cycles that finish shorter than the profile's
|
|
# (drifted-up) average and whose terminal pump-out lands
|
|
# *before* the drop into ENDING, so no in-ENDING end-spike
|
|
# ever arms - without it they hang to the fallback timeout
|
|
# (~30-44 min late) and their label can even drift to a longer
|
|
# near-duplicate profile. It is gated on
|
|
# ``current_duration >= expected`` so it can NOT fire during a
|
|
# long passive-drying phase that precedes a genuinely-late
|
|
# pump-out (e.g. an ECO cycle quiet from 50%-99% of expected):
|
|
# while still short of expected the cycle keeps waiting, and a
|
|
# real pump-out at ~99% arms the end-spike first. Takes the
|
|
# SOONER of the two anchors, so it can only ever shorten the
|
|
# wait, never extend it.
|
|
_spike_wait = self._dishwasher_end_spike_wait_s()
|
|
past_wait_period = current_duration >= (
|
|
self._expected_duration + _spike_wait
|
|
) or (
|
|
current_duration >= self._expected_duration
|
|
and self._time_below_threshold_gapfree
|
|
>= self._dishwasher_quiet_release_s()
|
|
)
|
|
if (
|
|
self._config.device_type == "dishwasher"
|
|
and not end_spike_seen
|
|
and not past_wait_period
|
|
):
|
|
self._logger.debug(
|
|
"Waiting for end spike (duration %.0fs, "
|
|
"expected %.0fs + %.0fs wait)",
|
|
current_duration,
|
|
self._expected_duration,
|
|
_spike_wait,
|
|
)
|
|
return # Don't finish yet, wait for spike
|
|
|
|
self._logger.info(
|
|
"Smart Termination: Profile '%s' match confirmed (duration %.0fs, "
|
|
"conf %.2f, spike_seen=%s), ending.",
|
|
self._matched_profile,
|
|
current_duration,
|
|
getattr(self, "_last_match_confidence", 0.0),
|
|
end_spike_seen,
|
|
)
|
|
# Keep tail when smart terminating (matches profile
|
|
# duration), but only as far as the program can
|
|
# actually reach - see _keep_tail_cap (#424).
|
|
self._finish_cycle(
|
|
timestamp,
|
|
status="completed",
|
|
termination_reason=TerminationReason.SMART,
|
|
keep_tail=True,
|
|
tail_cap=self._keep_tail_cap(start_time),
|
|
)
|
|
return
|
|
|
|
# --- DURATION-ANCHORED HARD FINALIZE (backstop) ---
|
|
# Separate safety net for a matched cycle whose Smart
|
|
# Termination is blocked (ambiguous / prefix-ambiguous match)
|
|
# and whose fallback energy gate is held open by a low standby
|
|
# baseline: without this it sits in ENDING until the 8 h cap /
|
|
# zombie-kill (#296/#311). Fires only well past the expected
|
|
# duration AND after a long *continuous* sub-threshold span, so
|
|
# it can never truncate a longer program mismatched to a shorter
|
|
# profile (a real longer program has high-power phases that keep
|
|
# resetting the quiet timer) — asymmetric, shorten-only. Not
|
|
# for user-paused cycles.
|
|
required_quiet = max(
|
|
self._ending_hard_finalize_quiet_s(),
|
|
float(max(self._config.off_delay, self._config.min_off_gap)),
|
|
)
|
|
# Require the required_quiet tail to be actually SAMPLED (no
|
|
# outage-sized gap): otherwise a telemetry outage that inflated
|
|
# _time_below_threshold could finalize an active cycle early.
|
|
# Walk in reverse so we can capture the boundary reading (the
|
|
# first reading outside the window) — an outage right before the
|
|
# window would be invisible if we only passed in-window timestamps.
|
|
quiet_ts: list[datetime] = []
|
|
_boundary_q: datetime | None = None
|
|
for _ts, _ in reversed(self._power_readings):
|
|
if (timestamp - _ts).total_seconds() <= required_quiet:
|
|
quiet_ts.append(_ts)
|
|
elif quiet_ts:
|
|
_boundary_q = _ts
|
|
break
|
|
if _boundary_q is not None:
|
|
quiet_ts.append(_boundary_q)
|
|
if (
|
|
self._expected_duration > 0
|
|
and current_duration
|
|
>= self._expected_duration * ENDING_HARD_FINALIZE_RATIO
|
|
and self._time_below_threshold >= required_quiet
|
|
and not self._verified_pause
|
|
and not self._window_has_outage_gap(quiet_ts)
|
|
):
|
|
self._logger.info(
|
|
"Duration-anchored finalize: cycle in ENDING at %.0fs "
|
|
"(%.1fx expected %.0fs), quiet %.0fs - Smart Termination "
|
|
"was blocked (ambiguous=%s); finalizing.",
|
|
current_duration,
|
|
current_duration / self._expected_duration,
|
|
self._expected_duration,
|
|
self._time_below_threshold,
|
|
self._match_ambiguous,
|
|
)
|
|
self._finish_cycle(
|
|
timestamp,
|
|
status="completed",
|
|
termination_reason=TerminationReason.TIMEOUT,
|
|
keep_tail=False,
|
|
)
|
|
return
|
|
|
|
# --- FALLBACK TIMEOUT CHECK ---
|
|
# Rule: To separate cycles, we must wait at least min_off_gap.
|
|
effective_off_delay = self._hazard_wait(
|
|
timestamp, max(self._config.off_delay, self._config.min_off_gap)
|
|
)
|
|
|
|
# Progress-aware shortening (register item 306). `min_off_gap` is
|
|
# there to bridge mid-cycle soak periods; once the run is past the
|
|
# matched programme's OWN expected length there is no soak left to
|
|
# bridge, so continuing to wait out a blind per-device prior just
|
|
# reports the end late.
|
|
#
|
|
# Measured on `devtools/end_gate_eval.py`, which IS committed.
|
|
# **`cycle_data/` is not**, so every n below depends on the local
|
|
# corpus and two different counts get quoted: cycles REPLAYED,
|
|
# and the subset that reached an ENDING exit, which is the only
|
|
# one that yields a lag and is what the harness table's `n`
|
|
# column reports. Current corpus: 273 replayed, 263 measured.
|
|
# Figures attributed below to "211/221" predate the #445 / #427 /
|
|
# #424 reporter exports being added mid-PR-#448, when the same
|
|
# corpus was 221 replayed / 211 measured. They are the same
|
|
# harness over a smaller corpus, not a drift.
|
|
#
|
|
# Re-cut on the current corpus (`--no-shortening` vs shipped,
|
|
# paired over 273 cycles): median end lag 13.43 -> 12.00 min
|
|
# overall and 27.50 -> 26.00 for washing machines, dishwasher p90
|
|
# 33.50 -> 19.70, and early ends (1.14% / 0.00%) and splits
|
|
# (2.66%) **identical in every scope**. It moves 32 of 273
|
|
# cycles: it buys little because it reaches few, and it costs
|
|
# nothing. Item 306's original figures (427 cycles, washing
|
|
# machines 12.9 -> 7.5 min, splits 3.75%) came from a harness
|
|
# that was never committed and **do not reproduce**; the safety
|
|
# half reproduces exactly. Treat the 427-cycle numbers as
|
|
# unverified; register item 329 holds that reconciliation, taken
|
|
# on the 211-measured corpus. Re-cut anything new with
|
|
# `end_gate_eval.py` rather than a throwaway script.
|
|
#
|
|
# Asymmetric and bounded, in the same spirit as _keep_tail_cap: it
|
|
# can only ever shorten, keeps the user's explicit `off_delay` as
|
|
# the floor (only the blind prior shrinks), and is inert when
|
|
# nothing matched. The bar it waits for is DEVICE-RESOLVED, not a
|
|
# fixed 1.05x (register item 355): `resolve_end_gate_late_ratio`
|
|
# returns 0.90 for `washing_machine` / `washer_dryer` and
|
|
# `END_GATE_LATE_RATIO` (1.05) for everything else - so on a washer
|
|
# the shortening starts BEFORE the expected end. That is the point:
|
|
# a washer's programme is load-adaptive, so a run sits below its
|
|
# profile mean about half the time by definition and the median
|
|
# washer reaches only 0.83 of a 1.05 bar, which put the rule out of
|
|
# reach for the device type that needed it most. Early ends stayed
|
|
# at 0.00% for washers at every ratio measured; the dishwashers,
|
|
# which do produce early ends below 1.0, keep 1.05.
|
|
# Gated on the SAME guard Smart Termination respects. The rule
|
|
# keys on `_expected_duration`, so it must not fire while the
|
|
# matcher says that duration is in doubt: an ambiguous match may
|
|
# mean a much longer look-alike is still plausible, and then "past
|
|
# the expected end" may really be "mid-soak in a longer programme".
|
|
# Without this the fallback timeout walks straight through and
|
|
# re-opens the #288 split-cycle bug - caught by
|
|
# test_smart_termination_blocked_by_ambiguous_match, where a 450 s
|
|
# soak dip sits right at the short profile's end.
|
|
# NOT also gated on `_last_match_confidence >=
|
|
# match_confidence_threshold`, and that is deliberate, not an
|
|
# oversight - the paragraph above says "the SAME guard Smart
|
|
# Termination respects" and means the ambiguity flag.
|
|
# Smart Termination does check confidence, because it ENDS a cycle
|
|
# early on a prediction; this rule only shortens a wait that is
|
|
# already past the programme's own expected end, so the asymmetry
|
|
# is intended. Adding the check was tried (PR #448 round 6) and
|
|
# measured on `devtools/end_gate_eval.py` over 221 replayed real
|
|
# cycles (211 of them measurable, the pre-reporter-export corpus
|
|
# described above): it moves **2 of them**, delaying one by 10.5
|
|
# min and one by 21 min, while early ends (1.42% / 0.00% at the >1
|
|
# and >5 min marks) and splits (3.32%) stay **exactly** where they
|
|
# were. It
|
|
# prevented no split and no early end - pure cost, so it was
|
|
# reverted. The corpus carries only 4 matched cycles under 0.4
|
|
# confidence, so it cannot prove the guard harmless either; the
|
|
# exposure is real (cycle `7c4598310016` is a 3h37m wash matched at
|
|
# 0.372 to a profile named "1:07", i.e. running the shortened gate
|
|
# for over two hours) and it still did not split. Re-run the
|
|
# harness before re-litigating this.
|
|
if (
|
|
self._matched_profile
|
|
and self._expected_duration > 0
|
|
and self._current_cycle_start is not None
|
|
):
|
|
_elapsed = self._gate_elapsed_s(timestamp) # item 511
|
|
# The bar this run has to clear. Normally the matched
|
|
# programme's own expected end; while the matcher still thinks
|
|
# a materially LONGER programme is plausible, that longer one's
|
|
# end instead (register item 330).
|
|
#
|
|
# Blocking outright on the ambiguity flags - which is what
|
|
# this did until item 330 - was costing almost every cycle the
|
|
# shortening. Measured on `devtools/end_gate_eval.py`: of the
|
|
# 94 cycles that ever pass 1.05x their own expected duration,
|
|
# **84 (89%) were blocked by an ambiguity flag**, so the rule
|
|
# reached 4.7% of cycles (10 of the 211 measurable on the
|
|
# pre-reporter-export corpus) and the median cycle still
|
|
# waited out the full `min_off_gap`.
|
|
#
|
|
# The flags were not wrong, they were too coarse. They exist to
|
|
# protect `_expected_duration` against "this is really a prefix
|
|
# of something longer" (#288 / #364) - a statement about
|
|
# DURATION, not about which label wins. (Only `_match_ambiguous`
|
|
# is left: the #364 prefix-fit flag was removed in 0.5.8.)
|
|
# Two programmes that score within the ambiguity margin and
|
|
# run the same length leave "past the expected end" true
|
|
# either way. So instead of
|
|
# refusing, raise the bar to the longest duration still in
|
|
# play: past THAT, no candidate is left for this to be a
|
|
# mid-soak of, which is exactly the condition the guard was
|
|
# standing in for.
|
|
#
|
|
# Absent information keeps the OLD refusal. A caller that does
|
|
# not send element 12 (an older Playground, most tests, any
|
|
# short tuple) leaves `_longest_candidate_duration` at 0.0,
|
|
# and an ambiguous match with no candidate durations to
|
|
# compare must block exactly as it did before - otherwise the
|
|
# #288 split-cycle reproduction
|
|
# `test_smart_termination_blocked_by_ambiguous_match` walks
|
|
# straight through, which is how the first draft of this was
|
|
# caught.
|
|
#
|
|
# The bar is `_fallback_shortening_bar` (one implementation
|
|
# with the item-469b ENDING hold): None, i.e. blocked, for an
|
|
# ambiguous match with no candidate durations. The bound
|
|
# itself is derived from the FULL pre-collapse candidate
|
|
# population, so a longer candidate ranked sixth or lower
|
|
# cannot hide from the bar - which it could when this read
|
|
# `MatchResult.candidates`, i.e. `candidates[:5]`.
|
|
_bar_s = self._fallback_shortening_bar(
|
|
self._expected_duration,
|
|
self._match_ambiguous,
|
|
self._longest_candidate_duration,
|
|
)
|
|
# Device-resolved (register item 355): 1.05 is out of reach
|
|
# for a load-adaptive washer, which reaches a median 0.83 of
|
|
# its EXPECTED duration before it stops.
|
|
#
|
|
# **But never against a RAISED bar.** When the match is
|
|
# ambiguous and a longer candidate exists, `_bar` is no longer
|
|
# the expected duration - it is the longest plausible
|
|
# programme, and it was raised precisely to say "a much longer
|
|
# look-alike is still on the table, so past the expected end
|
|
# does not mean done". Discounting that by 0.90 would shorten
|
|
# the wait 10% BEFORE the candidate it represents could even
|
|
# finish, and on the shipped washer defaults that drops the
|
|
# wait from `min_off_gap` to `max(off_delay, 300)` - one quiet
|
|
# interval away from finalising mid-programme and recording the
|
|
# rest as a second cycle, which is #288. The item-355 measurement
|
|
# was taken against the expected duration and says nothing about
|
|
# this case, so a raised bar keeps the original 1.05.
|
|
if _bar_s is not None and _elapsed >= _bar_s:
|
|
effective_off_delay = max(
|
|
self._config.off_delay,
|
|
min(self._config.min_off_gap, END_GATE_LATE_SECONDS),
|
|
)
|
|
|
|
# Energy gate always looks back off_delay seconds by default;
|
|
# overridden below for the dishwasher cap case so the window
|
|
# is consistent with the shortened effective_off_delay.
|
|
gate_window = self._config.off_delay
|
|
|
|
# Dishwasher-specific: after a terminal end spike (pump-out), an
|
|
# unmatched cycle doesn't need to wait the full min_off_gap (up to
|
|
# 9000s) before closing. Cap at 30 min so cycle 3 ends cleanly
|
|
# ~30 min after the pump-out rather than sitting open for hours.
|
|
if (
|
|
self._config.device_type == "dishwasher"
|
|
and not self._matched_profile
|
|
and self._end_spike_seen
|
|
):
|
|
effective_off_delay = min(effective_off_delay, 1800)
|
|
gate_window = effective_off_delay
|
|
|
|
# Opt-in terminal-drop fast finalize (asymmetric, shorten-only):
|
|
# a hard cliff-to-~0 sustained for TERMINAL_DROP_OFF_DELAY_SECONDS
|
|
# that began earlier than this device has ever legitimately gone
|
|
# quiet is almost certainly a real stop (plug pulled / cancelled),
|
|
# not a soak. Finalize now instead of waiting out the full
|
|
# soak-bridging min_off_gap. Only consulted when there is a longer
|
|
# wait to shorten and the provider is wired (ML/anomaly opt-in);
|
|
# the energy/defer gates are bypassed because the sustained sub-
|
|
# threshold span already proves the appliance is off, and the
|
|
# anomaly check has ruled out a legitimate early pause.
|
|
if (
|
|
self._terminal_drop_provider is not None
|
|
and not self._verified_pause
|
|
and effective_off_delay > TERMINAL_DROP_OFF_DELAY_SECONDS
|
|
and self._time_below_threshold >= TERMINAL_DROP_OFF_DELAY_SECONDS
|
|
and self._is_terminal_drop()
|
|
):
|
|
start_time = self._current_cycle_start or timestamp
|
|
current_duration = (timestamp - start_time).total_seconds()
|
|
self._logger.info(
|
|
"Terminal drop: anomalously-early power cliff after %.0fs "
|
|
"(device never quiet this early) - finalizing without the "
|
|
"full %.0fs soak wait.",
|
|
current_duration,
|
|
effective_off_delay,
|
|
)
|
|
self._finish_cycle(
|
|
timestamp,
|
|
status="interrupted",
|
|
termination_reason=TerminationReason.TERMINAL_DROP,
|
|
keep_tail=False,
|
|
)
|
|
return
|
|
|
|
if self._time_below_threshold >= effective_off_delay:
|
|
|
|
# Walk from the tail — readings are chronological so we can
|
|
# break as soon as we exceed the gate window (O(window) not O(n)).
|
|
recent_window = []
|
|
for r in reversed(self._power_readings):
|
|
if (timestamp - r[0]).total_seconds() <= gate_window:
|
|
recent_window.append(r)
|
|
else:
|
|
break
|
|
recent_window.reverse()
|
|
|
|
if not recent_window:
|
|
# Check deferred finish for matched profiles
|
|
start_time = self._current_cycle_start or timestamp
|
|
current_duration = self._gate_elapsed_s(timestamp) # item 511
|
|
|
|
if self._should_defer_finish(current_duration):
|
|
return
|
|
|
|
# For dishwashers, use the timeout timestamp as end_time
|
|
# (keep_tail=True) so that the stored cycle duration includes
|
|
# the passive drying phase. Without this, end_time snaps back
|
|
# to _last_active_time which may be set by a terminal drain
|
|
# spike mid-ENDING, producing a falsely short cycle duration.
|
|
keep_tail = self._config.device_type == "dishwasher"
|
|
self._finish_cycle(
|
|
timestamp,
|
|
status="completed",
|
|
keep_tail=keep_tail,
|
|
tail_cap=self._keep_tail_cap(start_time),
|
|
)
|
|
return
|
|
|
|
# Compute energy in recent window
|
|
recent_ts = np.array([r[0].timestamp() for r in recent_window])
|
|
recent_p = np.array([r[1] for r in recent_window])
|
|
max_gap_s = energy_gap_threshold_s(recent_ts)
|
|
recent_e = integrate_wh(recent_ts, recent_p, max_gap_s=max_gap_s)
|
|
|
|
energy_ok = recent_e <= self.config.end_energy_threshold
|
|
if not energy_ok and self._flat_standby_tail(timestamp):
|
|
# DETECT-08: the window's energy is a flat standby, not
|
|
# activity. Never before the full un-shortened wait, so a
|
|
# flat sub-stop phase the item-306 / hazard shortening
|
|
# would otherwise cut is still bridged by min_off_gap;
|
|
# far under stop, not before 2 h (see the constants).
|
|
self._logger.debug(
|
|
"Energy gate skipped: %.4fWh is a flat standby "
|
|
"after %.0fs quiet",
|
|
recent_e,
|
|
self._time_below_threshold,
|
|
)
|
|
energy_ok = True
|
|
if energy_ok:
|
|
start_time = self._current_cycle_start or timestamp
|
|
current_duration = self._gate_elapsed_s(timestamp) # item 511
|
|
|
|
if self._should_defer_finish(current_duration):
|
|
return
|
|
|
|
keep_tail = self._config.device_type == "dishwasher"
|
|
self._finish_cycle(
|
|
timestamp,
|
|
status="completed",
|
|
keep_tail=keep_tail,
|
|
tail_cap=self._keep_tail_cap(start_time),
|
|
)
|
|
else:
|
|
|
|
self._logger.debug(
|
|
"Cycle ending prevented by energy gate: %.4fWh > %.4fWh",
|
|
recent_e,
|
|
self._config.end_energy_threshold,
|
|
)
|
|
|
|
def _record_preroll(self, power: float, timestamp: datetime) -> None:
|
|
"""Buffer a reading seen before a cycle commits (#430).
|
|
|
|
Only while no cycle is open - once RUNNING, ``_power_readings`` is the
|
|
curve and this buffer would just duplicate it. STARTING counts as "not
|
|
open": a probe in STARTING may still abort, and those are precisely the
|
|
readings worth keeping.
|
|
|
|
Bounded by ``curve_preroll_seconds`` (itself capped at
|
|
``CURVE_PREROLL_MAX_SECONDS``), so the buffer holds seconds of data, not
|
|
an unbounded history.
|
|
"""
|
|
window = effective_curve_preroll_seconds(self._config.curve_preroll_seconds)
|
|
if window <= 0:
|
|
# Option off: keep the buffer empty rather than paying to fill one
|
|
# nothing will read, and so that enabling it mid-run cannot splice in
|
|
# readings from before the option was turned on.
|
|
if self._preroll_buffer:
|
|
self._preroll_buffer = []
|
|
return
|
|
if self._state not in (
|
|
STATE_OFF,
|
|
STATE_STARTING,
|
|
STATE_DELAY_WAIT,
|
|
STATE_UNKNOWN,
|
|
) and not (
|
|
# Item 515: a false start out of a terminal state returns there, not to
|
|
# OFF; it keeps recording as OFF did, so the next probe can still chain
|
|
# it. The buffer is empty there until a probe (the cycle end reset it).
|
|
self._preroll_buffer
|
|
and self._state in (STATE_FINISHED, STATE_INTERRUPTED, STATE_FORCE_STOPPED)
|
|
):
|
|
return
|
|
self._preroll_buffer.append((timestamp, float(power)))
|
|
cutoff = timestamp - timedelta(seconds=window)
|
|
# Readings arrive in order, so the stale prefix is contiguous.
|
|
drop = 0
|
|
for ts, _p in self._preroll_buffer:
|
|
if ts < cutoff:
|
|
drop += 1
|
|
else:
|
|
break
|
|
if drop:
|
|
del self._preroll_buffer[:drop]
|
|
|
|
def _preroll_for_commit(
|
|
self, timestamp: datetime, power: float
|
|
) -> list[tuple[datetime, float]]:
|
|
"""Readings to prepend to a cycle committing at ``timestamp`` (#430).
|
|
|
|
Walks the buffer backwards from the commit and stops at the first quiet
|
|
gap longer than ``PREROLL_CHAIN_BREAK_SECONDS`` - "the same start, probed
|
|
twice" rather than "an unrelated blip earlier". Then anchors on the
|
|
EARLIEST reading in that chain that is at or above ``start_threshold_w``:
|
|
anchoring on the window edge instead would drag standby into the curve
|
|
and move the cycle start to a moment the appliance was not yet doing
|
|
anything.
|
|
|
|
Returns [] whenever there is nothing to add, so the caller's fast path is
|
|
a single emptiness test.
|
|
"""
|
|
window = effective_curve_preroll_seconds(self._config.curve_preroll_seconds)
|
|
if window <= 0 or not self._preroll_buffer:
|
|
return []
|
|
|
|
# Everything strictly before this commit, most recent first.
|
|
prior = [(ts, p) for ts, p in self._preroll_buffer if ts < timestamp]
|
|
if not prior:
|
|
return []
|
|
|
|
chain: list[tuple[datetime, float]] = []
|
|
next_ts = timestamp
|
|
for ts, p in reversed(prior):
|
|
if (next_ts - ts).total_seconds() > PREROLL_CHAIN_BREAK_SECONDS:
|
|
break
|
|
chain.append((ts, p))
|
|
next_ts = ts
|
|
if not chain:
|
|
return []
|
|
chain.reverse() # chronological
|
|
|
|
threshold = float(self._config.start_threshold_w)
|
|
anchor = next(
|
|
(i for i, (_ts, p) in enumerate(chain) if p >= threshold), None
|
|
)
|
|
if anchor is None:
|
|
return [] # the chain is all standby - nothing of this cycle in it
|
|
return chain[anchor:]
|
|
|
|
def _apply_curve_preroll(self, timestamp: datetime, power: float) -> None:
|
|
"""Prepend buffered pre-commit readings to the freshly-started cycle (#430).
|
|
|
|
Moves ``_current_cycle_start`` back with them, and that is not optional:
|
|
the stored duration is ``end_time - _current_cycle_start`` while matching
|
|
resamples ``_power_readings``, so a curve that started earlier than the
|
|
pointer would describe a different run from the one whose duration is
|
|
recorded. The two existing back-anchors (the anti-wrinkle candidate
|
|
window and the DELAY_WAIT high-start anchor) move the pointer for exactly
|
|
the same reason.
|
|
|
|
**Record-only: this must never make a cycle easier to START.**
|
|
``_energy_since_idle_wh`` is deliberately left alone. Despite the name it
|
|
is not the cycle's energy - the stored figure is integrated from
|
|
``power_data`` at persistence, so it picks the pre-roll up for free - it
|
|
is the accumulator the STARTING -> RUNNING gate reads
|
|
(``>= start_energy_threshold``). Feeding it the pre-roll would let an
|
|
aborted probe's energy be re-spent on the next probe's start gate, so two
|
|
blips that each failed the gate could together pass it: exactly the
|
|
phantom cycle #403 was fixed to prevent. The same argument covers
|
|
``_time_above_threshold``, which is likewise untouched.
|
|
|
|
``_cycle_max_power`` IS updated, because that is a property of the run
|
|
being recorded (it gates the anti-crease path), not of admitting it.
|
|
"""
|
|
preroll = self._preroll_for_commit(timestamp, power)
|
|
if not preroll:
|
|
return
|
|
|
|
start_ts = preroll[0][0]
|
|
# Callers do not all commit with the pointer at ``timestamp``. The
|
|
# DELAY_WAIT confirmation has already back-anchored it to its first
|
|
# sustained-high reading, and a pre-roll window shorter than
|
|
# ``start_duration_threshold`` would otherwise move that pointer FORWARD
|
|
# and shorten the delayed start. So keep whatever the caller anchored
|
|
# before the chain, and only ever move the pointer earlier.
|
|
earlier = [(ts, p) for ts, p in self._power_readings if ts < start_ts]
|
|
self._power_readings = [*earlier, *preroll, (timestamp, power)]
|
|
self._current_cycle_start = min(
|
|
start_ts, self._current_cycle_start or start_ts
|
|
)
|
|
self._cycle_max_power = max(p for _ts, p in self._power_readings)
|
|
self._logger.debug(
|
|
"Curve pre-roll: carried %d reading(s) covering %.0fs from aborted "
|
|
"start probe(s) into this cycle (start moved back to %s).",
|
|
len(preroll),
|
|
(timestamp - start_ts).total_seconds(),
|
|
start_ts.isoformat(),
|
|
)
|
|
|
|
def _transition_to(self, new_state: str, timestamp: datetime) -> None:
|
|
"""Handle state transitions."""
|
|
if self._state == new_state:
|
|
return
|
|
|
|
old_state = self._state
|
|
self._state = new_state
|
|
self._state_enter_time = timestamp
|
|
self._time_in_state = 0.0
|
|
self._sub_state = new_state.capitalize() # Default substate
|
|
|
|
# Bound each ENDING episode's ML-guard deferral independently: clear the
|
|
# tracker whenever we are not in ENDING (e.g. on resume back to RUNNING).
|
|
if new_state != STATE_ENDING:
|
|
self._ml_defer_start_duration = None
|
|
|
|
# Item 501: hiding is per probe, and a re-probe only follows a false start
|
|
# straight back in OFF (any other state means the standby run is over).
|
|
if new_state != STATE_STARTING:
|
|
self._probe_hidden = False
|
|
self._probe_wait_since = None # item 504: per probe, like the hiding
|
|
self._probe_terminal_from = None # item 515, likewise
|
|
if new_state not in (STATE_OFF, STATE_STARTING):
|
|
self._standby_reprobe = False
|
|
if new_state not in (STATE_RUNNING, STATE_PAUSED, STATE_ENDING):
|
|
self._clear_stall() # #452: a stall belongs to one open cycle
|
|
self._stall_spans = [] # item 511
|
|
self._user_pause_spans = [] # item 514
|
|
self._user_pause_since = None
|
|
|
|
# Reset energy accumulator on transition to OFF
|
|
if new_state == STATE_OFF:
|
|
self._energy_since_idle_wh = 0.0
|
|
# Also reset idle time tracker when leaving ANTI_WRINKLE
|
|
self._anti_wrinkle_idle_time = 0.0
|
|
if not self._preserve_delay_band_on_off:
|
|
self._delay_band_start = None
|
|
self._delay_band_seconds = 0.0
|
|
self._delay_band_peak = 0.0
|
|
self._delay_wait_true_off_seconds = 0.0
|
|
self._delay_wait_high_start = None
|
|
self._delay_wait_high_power = None
|
|
self._preserve_delay_band_on_off = False
|
|
# Clear the paused-STARTING true-off accumulator so a later STARTING
|
|
# cycle cannot inherit stale hold time and finalize to OFF prematurely
|
|
# (this path is also reached via the paused-STARTING cancellation).
|
|
self._starting_paused_off_since = None
|
|
|
|
# Reset end spike tracker when entering ENDING state
|
|
if new_state == STATE_ENDING:
|
|
self._end_spike_seen = False
|
|
self._end_spike_duration = 0.0
|
|
elif new_state == STATE_DELAY_WAIT:
|
|
# Band-accumulation tracker already played its role getting us
|
|
# here; reset it so a future OFF→band cycle starts fresh.
|
|
self._delay_band_start = None
|
|
self._delay_band_seconds = 0.0
|
|
self._delay_band_peak = 0.0
|
|
self._delay_wait_true_off_seconds = 0.0
|
|
self._delay_wait_high_start = None
|
|
self._delay_wait_high_power = None
|
|
self._sub_state = "Waiting to Start"
|
|
self._preserve_delay_band_on_off = False
|
|
# Like OFF: a false start returning here (item 504) leaves no energy.
|
|
self._energy_since_idle_wh = 0.0
|
|
elif new_state == STATE_ANTI_WRINKLE:
|
|
self._anti_wrinkle_candidate_start = None
|
|
self._anti_wrinkle_candidate_peak = 0.0
|
|
self._anti_wrinkle_candidate_start_power = 0.0
|
|
self._anti_wrinkle_idle_time = 0.0 # Reset idle time when entering ANTI_WRINKLE
|
|
self._sub_state = "Anti-Wrinkle"
|
|
elif new_state == STATE_STARTING:
|
|
# Reset idle time if exiting ANTI_WRINKLE to STARTING (high-power burst resumed)
|
|
self._anti_wrinkle_idle_time = 0.0
|
|
# Fresh STARTING cycle: never inherit a prior cycle's true-off hold.
|
|
self._starting_paused_off_since = None
|
|
elif new_state == STATE_RUNNING:
|
|
self._delay_band_start = None
|
|
self._delay_band_seconds = 0.0
|
|
self._delay_band_peak = 0.0
|
|
self._preserve_delay_band_on_off = False
|
|
|
|
self._logger.debug("Transition: %s -> %s at %s", old_state, new_state, timestamp)
|
|
self._on_state_change(old_state, new_state)
|
|
|
|
def _ml_end_confidence(self) -> float | None:
|
|
"""P(the current low-power event is the true end) from the opt-in ML guard.
|
|
|
|
Builds the offset-second trace from the current cycle's readings and asks
|
|
the injected provider. Returns None when there is no provider, no cycle
|
|
start, or the provider declines (ML off / unmatched / model unavailable),
|
|
so the caller keeps the existing power/energy-based behavior.
|
|
"""
|
|
provider = self._end_confidence_provider
|
|
start = self._current_cycle_start
|
|
if provider is None or start is None or not self._power_readings:
|
|
return None
|
|
# Throttle: reuse the last result within the recompute window, but only when
|
|
# it was computed for THIS cycle and the same expected_duration (which can
|
|
# change under overrun) — otherwise recompute.
|
|
now_ts = self._power_readings[-1][0]
|
|
exp = float(self._expected_duration)
|
|
cache = self._ml_end_cache
|
|
if (
|
|
cache is not None
|
|
and cache[1] == exp
|
|
and cache[2] == start
|
|
and (now_ts - cache[0]).total_seconds() < ML_PROVIDER_THROTTLE_SECONDS
|
|
):
|
|
return cache[3]
|
|
points = [
|
|
((ts - start).total_seconds(), float(power))
|
|
for ts, power in self._power_readings
|
|
]
|
|
try:
|
|
result = provider(points, exp)
|
|
except Exception: # noqa: BLE001 - ML must never break detection
|
|
result = None
|
|
self._ml_end_cache = (now_ts, exp, start, result)
|
|
return result
|
|
|
|
def _flat_standby_tail(self, timestamp: datetime) -> bool:
|
|
"""Whether the ENDING quiet is a flat standby the energy gate must not pin.
|
|
|
|
DETECT-08. A sampled (no outage hole) window of the last
|
|
max(off_delay, STANDBY_BAND_WINDOW_S) seconds whose readings are all above
|
|
0 W and within max(ENDING_FLAT_STANDBY_SPREAD_FLOOR_W,
|
|
ENDING_FLAT_STANDBY_SPREAD_FRAC x the highest) of each other, after
|
|
max(off_delay, min_off_gap) of quiet when every reading is at least
|
|
ENDING_FLAT_STANDBY_NEAR_STOP_FRAC x stop_threshold_w, else after
|
|
ENDING_FLAT_STANDBY_LOOSE_QUIET_S. A window reaching back past the start
|
|
of the quiet holds an above-stop reading and so is not flat.
|
|
"""
|
|
quiet = self._time_below_threshold
|
|
base_wait = float(max(self._config.off_delay, self._config.min_off_gap))
|
|
if quiet < base_wait:
|
|
return False
|
|
span = max(float(self._config.off_delay), STANDBY_BAND_WINDOW_S)
|
|
powers: list[float] = []
|
|
window_ts: list[datetime] = []
|
|
boundary: datetime | None = None
|
|
for ts, p in reversed(self._power_readings):
|
|
if (timestamp - ts).total_seconds() <= span:
|
|
powers.append(float(p))
|
|
window_ts.append(ts)
|
|
else:
|
|
boundary = ts
|
|
break
|
|
if boundary is None or len(powers) < ENDING_FLAT_STANDBY_MIN_READINGS:
|
|
return False
|
|
lo, hi = min(powers), max(powers)
|
|
if lo <= 0.0:
|
|
return False
|
|
if (hi - lo) > max(
|
|
ENDING_FLAT_STANDBY_SPREAD_FLOOR_W, ENDING_FLAT_STANDBY_SPREAD_FRAC * hi
|
|
):
|
|
return False
|
|
near_stop = lo >= ENDING_FLAT_STANDBY_NEAR_STOP_FRAC * float(
|
|
self._config.stop_threshold_w
|
|
)
|
|
if not near_stop and quiet < max(base_wait, ENDING_FLAT_STANDBY_LOOSE_QUIET_S):
|
|
return False
|
|
return not self._window_has_outage_gap([boundary, *window_ts])
|
|
|
|
def _window_has_outage_gap(self, window_ts: list[datetime]) -> bool:
|
|
"""Whether a 'sustained window' contains a data-outage-sized hole.
|
|
|
|
The span + coverage checks in the standby / anti-crease window scans accept
|
|
e.g. three readings spanning the window even if a long unobserved gap sits
|
|
between them (a sensor dropout, or a sparse burst next to one old reading).
|
|
Finalizing on such a window could wrongly cut an active cycle, so reject it.
|
|
The gap ceiling is the sensor's own data-driven outage threshold
|
|
(``energy_gap_threshold_s`` over the full trace), so a change-only sensor's
|
|
legitimately-sparse stable stretches (tens of seconds between reports) are
|
|
NOT rejected while a genuine dropout is.
|
|
"""
|
|
if len(window_ts) < 2:
|
|
return True # too few points to trust as a sustained window
|
|
max_gap = self._outage_threshold_s()
|
|
ordered = sorted(t.timestamp() for t in window_ts)
|
|
return any((b - a) > max_gap for a, b in itertools.pairwise(ordered))
|
|
|
|
def _outage_threshold_s(self) -> float:
|
|
"""Sensor-adaptive gap ceiling (seconds): intervals longer than this are
|
|
treated as telemetry outages, not observed quiet. Data-driven from the
|
|
trace's own cadence (`energy_gap_threshold_s`), so a change-only sensor's
|
|
sparse-but-real stable stretches are not mistaken for a dropout.
|
|
"""
|
|
all_ts = np.array(
|
|
[r[0].timestamp() for r in self._power_readings], dtype=float
|
|
)
|
|
return energy_gap_threshold_s(all_ts)
|
|
|
|
def _standby_band_plateau(self, timestamp: datetime) -> float | None:
|
|
"""The stuck plateau's highest reading, or None when the cycle is not stuck.
|
|
|
|
Whether a RUNNING cycle is stuck on a flat standby plateau (#296).
|
|
|
|
Returns True only when ALL of the following hold, so this can never end
|
|
an active low-power phase:
|
|
|
|
* the device is a wet appliance where a stuck baseline is unambiguously
|
|
anomalous (``STANDBY_BAND_FINALIZE_DEVICE_TYPES``);
|
|
* a profile is matched and elapsed >= ``STANDBY_BAND_MIN_RATIO`` x the
|
|
expected duration (well past when it should have ended);
|
|
* not user-paused;
|
|
* the most recent >= ``STANDBY_BAND_WINDOW_S`` of readings are ALL at or
|
|
below ``STANDBY_BAND_MAX_FRACTION`` of the cycle's own peak power AND
|
|
span no more than ``STANDBY_BAND_FLATNESS_FRACTION`` of the peak (a flat
|
|
plateau, not fluctuating activity).
|
|
|
|
The expensive window scan runs only after the cheap duration gate passes,
|
|
so normal cycles never pay for it.
|
|
"""
|
|
if self._config.device_type not in STANDBY_BAND_FINALIZE_DEVICE_TYPES:
|
|
return None
|
|
if getattr(self, "_verified_pause", False):
|
|
return None
|
|
if not (self._matched_profile and self._expected_duration > 0):
|
|
return None
|
|
start = self._current_cycle_start
|
|
if start is None:
|
|
return None
|
|
current_duration = self._gate_elapsed_s(timestamp) # item 511
|
|
if current_duration < self._expected_duration * STANDBY_BAND_MIN_RATIO:
|
|
return None
|
|
# #399 interaction, load-bearing since the gate above dropped from 2.0x to
|
|
# 1.0x expected (#445): a washer can sit quiet below anti_wrinkle_max_power
|
|
# for minutes BEFORE its final spin, and that quiet is a flat sub-10%-of-peak
|
|
# plateau like any other. Finalising there is exactly the failure #399 fixed
|
|
# - the spin then arrives and opens a second cycle record. Defer while the
|
|
# matched profile still owes this run its terminal high-power block. Shares
|
|
# the predicate with the anti-crease finalise so the two release together,
|
|
# and it fails open on every missing input (no profile block, non-terminal
|
|
# block, past the ANTI_CREASE_SPIN_WAIT_MAX_RATIO cap), so an appliance that
|
|
# never spins - the #445 Miele, which has no terminal block at all - is not
|
|
# delayed by it.
|
|
#
|
|
# **This used to be inert on the devices it exists for (register item
|
|
# 351).** `_anticrease_spin_pending` needs element 10, and the manager
|
|
# supplied it only when `anti_wrinkle_enabled` was true -
|
|
# `DEFAULT_ANTI_WRINKLE_ENABLED` is False, so on a washer the predicate
|
|
# returned False immediately and none of the above happened. Measured on
|
|
# the 273-cycle replay corpus: the band fired 14 times, 11 with the guard
|
|
# inert, and 6 of those 11 had a reading above `min_power` still ahead,
|
|
# i.e. would split. `terminal_high_for_guards` (module level in this file,
|
|
# shared by the manager's live match tuple and the Playground's sim tuple
|
|
# since round 31 - it used to be two hand-copies) now arms it for
|
|
# `STANDBY_BAND_FINALIZE_DEVICE_TYPES` against a share of the cycle's own
|
|
# peak - the same `STANDBY_BAND_MAX_FRACTION` used below, so the rule is
|
|
# "wait while the profile still owes a block above the plateau you are
|
|
# sitting on" and there is no new tunable. The bar travels WITH the block
|
|
# as element 4, because `_high_power_seconds_since` has to count live
|
|
# seconds above the same number.
|
|
#
|
|
# Measured end to end on `devtools/end_gate_eval.py`: washing-machine
|
|
# splits 4.49% -> 1.90%, median end lag 26.00 -> 24.74 min (it does not
|
|
# cost time - a cycle that used to split now finishes once), match rate
|
|
# 89.7% -> 92.4%, early ends unchanged at 0.00%, dishwashers identical.
|
|
# A ceiling of 0.15 caught the 6th split too but deferred 10 of the 11
|
|
# firings, and a deferral with no spin ahead waits out
|
|
# ANTI_CREASE_SPIN_WAIT_MAX_RATIO (1.25x expected, ~32 min on a 2:09
|
|
# wash), so it bought the last split for three long waits. 0.10 was the
|
|
# maintainer's call.
|
|
if self._anticrease_spin_pending(timestamp):
|
|
return None
|
|
peak = float(self._cycle_max_power)
|
|
if peak <= 0:
|
|
return None
|
|
|
|
level_ceiling = peak * STANDBY_BAND_MAX_FRACTION
|
|
# Walk the tail; readings are chronological so we can break once outside
|
|
# the window (O(window), not O(n)).
|
|
window: list[float] = []
|
|
window_ts: list[datetime] = []
|
|
oldest_in_window: datetime | None = None
|
|
saw_older = False # a reading older than the window exists -> full coverage
|
|
_standby_boundary_ts: datetime | None = None
|
|
for ts, p in reversed(self._power_readings):
|
|
if (timestamp - ts).total_seconds() <= STANDBY_BAND_WINDOW_S:
|
|
window.append(float(p))
|
|
window_ts.append(ts)
|
|
oldest_in_window = ts
|
|
else:
|
|
saw_older = True
|
|
_standby_boundary_ts = ts # boundary: last reading before the window
|
|
break
|
|
# The plateau must actually SPAN the required window (data exists from
|
|
# before it), not just a couple of recent samples, and have enough points
|
|
# to judge. `saw_older` (rather than an exact span >= WINDOW check) is
|
|
# robust to sample phase/granularity: with e.g. 30 s sampling the oldest
|
|
# in-window reading is typically only ~570-599 s old, which an exact check
|
|
# would wrongly reject. A coverage sanity (oldest >= 90% of the window)
|
|
# plus an adjacent-gap check (``_window_has_outage_gap``) guard against a
|
|
# sparse burst of samples sitting next to one old reading across a dropout.
|
|
if (
|
|
oldest_in_window is None
|
|
or not saw_older
|
|
or len(window) < 3
|
|
or (timestamp - oldest_in_window).total_seconds()
|
|
< STANDBY_BAND_WINDOW_S * 0.9
|
|
or self._window_has_outage_gap(
|
|
[_standby_boundary_ts, *window_ts]
|
|
if _standby_boundary_ts is not None
|
|
else window_ts
|
|
)
|
|
):
|
|
return None
|
|
hi = max(window)
|
|
lo = min(window)
|
|
if hi > level_ceiling:
|
|
return None # a real active reading in the window - not standby
|
|
flatness_limit = max(
|
|
STANDBY_BAND_FLATNESS_FLOOR_W, peak * STANDBY_BAND_FLATNESS_FRACTION
|
|
)
|
|
if (hi - lo) > flatness_limit:
|
|
return None # fluctuating - still doing work
|
|
# Two tiers. At or just above the stop threshold is the #445 shape - an
|
|
# appliance whose standby the end gates cannot see - and it closes at the
|
|
# expected duration. Anything else that passed the loose test above (a 0 W
|
|
# soak, a 60 W rinse; flat and under 10% of a heater peak) waits for twice
|
|
# the expected duration, as it did before 0.5.7. See the constants.
|
|
stop = float(self._config.stop_threshold_w)
|
|
near_stop_ceiling = standby_near_stop_ceiling(stop)
|
|
if (
|
|
stop > 0
|
|
and float(np.median(window)) >= stop
|
|
and hi <= near_stop_ceiling
|
|
# #452: a halt says the programme is not done (it stopped on its
|
|
# standby draw before its end), so the near-stop tier waits; the
|
|
# loose tier below still bounds it at STANDBY_BAND_LOOSE_MIN_RATIO x
|
|
# expected.
|
|
and not self._stall_holds_standby_band()
|
|
):
|
|
return hi
|
|
if current_duration >= self._expected_duration * STANDBY_BAND_LOOSE_MIN_RATIO:
|
|
return hi
|
|
return None
|
|
|
|
def _maybe_finalize_standby_band(self, timestamp: datetime, power: float) -> bool:
|
|
"""Finish the cycle when it is stuck on a standby plateau; True if it did.
|
|
|
|
Called from RUNNING on every reading and from ENDING on every above-stop
|
|
reading (DETECT-03); `_standby_band_plateau` holds every gate.
|
|
"""
|
|
plateau_hi = self._standby_band_plateau(timestamp)
|
|
if plateau_hi is None:
|
|
return False
|
|
start_time = self._current_cycle_start or timestamp
|
|
current_duration = (timestamp - start_time).total_seconds()
|
|
# The plateau sits ABOVE stop_threshold, so it keeps advancing
|
|
# _last_active_time and the default keep_tail=False trim would NOT
|
|
# remove it - inflating the stored duration/energy with minutes of
|
|
# standby. Snap the end back to the last reading above the PLATEAU
|
|
# and drop only the plateau run. This used to snap back to the last
|
|
# reading above 10% of the cycle's peak, which cut every lower-power
|
|
# phase after the last heating burst - 31 min of a Miele's rinse and
|
|
# spin in one replayed cycle.
|
|
trim_ceiling = plateau_hi + max(0.5, 0.1 * plateau_hi)
|
|
plateau_start_idx = None
|
|
for i in range(len(self._power_readings) - 1, -1, -1):
|
|
if float(self._power_readings[i][1]) > trim_ceiling:
|
|
plateau_start_idx = i
|
|
break
|
|
if (
|
|
plateau_start_idx is not None
|
|
and plateau_start_idx < len(self._power_readings) - 1
|
|
):
|
|
self._power_readings = self._power_readings[: plateau_start_idx + 1]
|
|
self._last_active_time = self._power_readings[-1][0]
|
|
current_duration = (self._last_active_time - start_time).total_seconds()
|
|
self._logger.info(
|
|
"Standby-band finalize (%s): flat plateau ~%.1fW (peak %.0fW) held "
|
|
"past expected %.0fs - appliance finished but holds a standby "
|
|
"draw above stop_threshold; finalizing (plateau trimmed, "
|
|
"duration %.0fs).",
|
|
self._state,
|
|
power,
|
|
self._cycle_max_power,
|
|
self._expected_duration,
|
|
current_duration,
|
|
)
|
|
self._finish_cycle(
|
|
timestamp,
|
|
status="completed",
|
|
termination_reason=TerminationReason.TIMEOUT,
|
|
keep_tail=False,
|
|
)
|
|
return True
|
|
|
|
def _anticrease_gate_open(self, timestamp: datetime) -> bool:
|
|
"""Core anti-crease gate (#296): everything except the current power level
|
|
and the low-power-window check. Shared by the match freeze
|
|
(``_in_anticrease_freeze``) and the finalise (``_is_anticrease_tail``).
|
|
|
|
True only when a genuinely energetic, confidently-matched cycle for an
|
|
anti-wrinkle device is PAST its expected duration - the discriminator that
|
|
separates the post-wash anti-crease tail from a mid-wash low-power trough
|
|
(a washer spends most of its cycle below ``anti_wrinkle_max_power``, but a
|
|
mid-wash trough is always BEFORE the expected duration, the tail after it).
|
|
|
|
That "past expected" fraction is per-appliance since #429
|
|
(``anti_crease_finalize_ratio``, default 0.98). **Lowering it trades away
|
|
exactly the guarantee in the paragraph above**, so it is meant for dryers
|
|
whose sensor-dry runtime follows the load and whose tumble tail would
|
|
otherwise sit until the fallback timeout. Nothing downstream can stand in
|
|
for it on a washer: the low-power window check cannot separate a trough
|
|
from a tail (both are below ``anti_wrinkle_max_power`` by definition),
|
|
``_smart_term_power_plausible`` compares the trailing mean against the
|
|
matched profile's OWN tail level, which is equally low, and
|
|
``_anticrease_spin_pending`` fails open when the profile carries no
|
|
terminal high block.
|
|
"""
|
|
if not self._config.anti_wrinkle_enabled:
|
|
return False
|
|
if self._config.device_type not in (
|
|
DEVICE_TYPE_WASHING_MACHINE,
|
|
DEVICE_TYPE_DRYER,
|
|
DEVICE_TYPE_WASHER_DRYER,
|
|
):
|
|
return False
|
|
if getattr(self, "_verified_pause", False):
|
|
return False
|
|
if not (self._matched_profile and self._expected_duration > 0):
|
|
return False
|
|
if self._last_match_confidence < self._config.match_confidence_threshold:
|
|
return False
|
|
# The #288 full-shape verdict only (the wider #364 prefix-fit flag never
|
|
# reached this gate, and was removed in 0.5.8): a false block here
|
|
# disables the finalise AND the match freeze, and because
|
|
# the tumble bursts recur faster than off_delay neither the fallback timeout
|
|
# nor ENDING_HARD_FINALIZE can close the cycle - that is the #296 hang.
|
|
if self._match_ambiguous or self._match_prefix_ambiguous_full_shape:
|
|
return False
|
|
if self._cycle_max_power <= float(self._config.anti_wrinkle_max_power):
|
|
return False # never a hot/energetic cycle - leave low-power programs alone
|
|
# Cheap clock test first, so the trailing-power scan below is skipped for the
|
|
# whole mid-wash phase (it only matters once we are past-expected).
|
|
start = self._current_cycle_start
|
|
if start is None:
|
|
return False
|
|
current_duration = self._gate_elapsed_s(timestamp) # programme time, item 511
|
|
# Held to the documented 0.50-1.00 range on READ, not at construction: the
|
|
# manager assigns this field directly on an options reload, and the value can
|
|
# arrive from an import or the Playground, neither of which range-checks it.
|
|
# A stored 0.0 would satisfy the test below for every duration and hand the
|
|
# gate a mid-wash trough.
|
|
finalize_ratio = effective_anticrease_finalize_ratio(
|
|
self._config.anti_crease_finalize_ratio
|
|
)
|
|
if current_duration < self._expected_duration * finalize_ratio:
|
|
return False
|
|
# #364: "past expected" only means "past the wash" when expected belongs to
|
|
# the RIGHT profile. A whole washer wash phase sits below
|
|
# anti_wrinkle_max_power, so with a mis-matched shorter profile this gate
|
|
# would open mid-wash. Requiring the trailing power to look like this
|
|
# profile's own tail restores the guarantee the ratio alone used to give.
|
|
if not self._smart_term_power_plausible(timestamp):
|
|
return False
|
|
return True
|
|
|
|
def _in_anticrease_freeze(self, timestamp: datetime) -> bool:
|
|
"""Whether match updates should be frozen (#296): the anti-crease gate is
|
|
open AND the most recent reading is in the low-power regime (at or below
|
|
``anti_wrinkle_max_power``).
|
|
|
|
Deliberately lighter than ``_is_anticrease_tail`` - it does NOT wait for the
|
|
full ``ANTI_CREASE_CONFIRM_WINDOW_S``, so the confident pre-tail match is
|
|
preserved from the instant the cycle crosses its expected duration in a
|
|
low-power state, before the window accrues. Without this a match that
|
|
degrades to ambiguous within the first window's worth of tail would
|
|
deadlock both the freeze and the finalise (both require an unambiguous
|
|
match). Self-correcting: a heating burst above ``anti_wrinkle_max_power``
|
|
leaves the regime and re-arms matching.
|
|
"""
|
|
if not self._power_readings:
|
|
return False
|
|
if float(self._power_readings[-1][1]) > float(
|
|
self._config.anti_wrinkle_max_power
|
|
):
|
|
return False
|
|
return self._anticrease_gate_open(timestamp)
|
|
|
|
def _is_anticrease_tail(self, timestamp: datetime) -> bool:
|
|
"""Whether a matched, past-expected cycle has settled into the anti-crease
|
|
tumble tail (#296) - the trigger for the finalise into STATE_ANTI_WRINKLE.
|
|
|
|
Miele-style "Knitterschutz": after the wash proper ends, the machine holds
|
|
a constant baseline plus periodic sub-``anti_wrinkle_max_power`` tumble
|
|
bursts (no heating) until the door is opened. Because those bursts recur
|
|
faster than off_delay they keep reviving the cycle out of ENDING, so the
|
|
normal power-off path never finalises it and STATE_ANTI_WRINKLE - which is
|
|
built to absorb the tail and split off the next wash - never engages.
|
|
Recognising the tail lets us finalise into anti-wrinkle directly.
|
|
|
|
Requires the core gate (``_anticrease_gate_open``) AND that the most recent
|
|
>= ``ANTI_CREASE_CONFIRM_WINDOW_S`` of readings are ALL at or below
|
|
``anti_wrinkle_max_power`` (we are in the low-power tail, clear of the final
|
|
spin and not mid-heating). The expensive window scan runs only after the
|
|
cheap gate passes, so normal cycles never pay for it.
|
|
"""
|
|
if not self._anticrease_gate_open(timestamp):
|
|
return False
|
|
# Item 511: a stalled wash is halted, not in its tumble tail (the #296
|
|
# bursts leave the stall band, so a real tail never shows as stalled). The
|
|
# standby band's loose tier still bounds the plateau.
|
|
if STALL_EXCLUDED_FROM_GATES and self._stall_active:
|
|
return False
|
|
# #399: only the finalise, never _anticrease_gate_open. A false block in the
|
|
# shared gate would also kill the match freeze, and because the tumble bursts
|
|
# recur faster than off_delay neither the fallback timeout nor
|
|
# ENDING_HARD_FINALIZE could then close the cycle - that is the #296 hang.
|
|
if self._anticrease_spin_pending(timestamp):
|
|
return False
|
|
max_power = float(self._config.anti_wrinkle_max_power)
|
|
# Walk the tail; readings are chronological so we can break once outside the
|
|
# window (O(window), not O(n)).
|
|
window: list[float] = []
|
|
window_ts: list[datetime] = []
|
|
oldest_in_window: datetime | None = None
|
|
saw_older = False
|
|
_ac_boundary_ts: datetime | None = None
|
|
for ts, p in reversed(self._power_readings):
|
|
if (timestamp - ts).total_seconds() <= ANTI_CREASE_CONFIRM_WINDOW_S:
|
|
window.append(float(p))
|
|
window_ts.append(ts)
|
|
oldest_in_window = ts
|
|
else:
|
|
saw_older = True
|
|
_ac_boundary_ts = ts # boundary: last reading before the window
|
|
break
|
|
# The low-power tail must actually SPAN the window (data exists from before
|
|
# it) and have enough points to judge - not just a couple of recent samples.
|
|
# ``saw_older`` plus a coverage sanity and an adjacent-gap check
|
|
# (``_window_has_outage_gap``) is robust to sample phase/granularity while
|
|
# rejecting a dropout-sized hole (mirrors _standby_band_plateau).
|
|
if (
|
|
oldest_in_window is None
|
|
or not saw_older
|
|
or len(window) < 3
|
|
or (timestamp - oldest_in_window).total_seconds()
|
|
< ANTI_CREASE_CONFIRM_WINDOW_S * 0.9
|
|
or self._window_has_outage_gap(
|
|
[_ac_boundary_ts, *window_ts]
|
|
if _ac_boundary_ts is not None
|
|
else window_ts
|
|
)
|
|
):
|
|
return False
|
|
if max(window) > max_power:
|
|
return False # a heating / high-spin reading in the window - still washing
|
|
return True
|
|
|
|
def _anticrease_spin_pending(self, timestamp: datetime) -> bool:
|
|
"""Whether the matched profile still owes this run a terminal high-power
|
|
event - i.e. the anti-crease finalise must wait (#399).
|
|
|
|
``_is_anticrease_tail``'s two conditions both look backwards: past expected,
|
|
and quiet for the confirm window. A programme whose final spin lands just
|
|
past 0.98 x expected, after a long sub-``anti_wrinkle_max_power`` rinse
|
|
stretch, satisfies both while the spin is still ahead - so the wash was
|
|
finalised into anti-wrinkle and the spin opened a SECOND cycle record.
|
|
|
|
The profile carries the missing information: where its own last high-power
|
|
block sits and how long it runs. If that block is terminal (starts at or
|
|
after ``ANTI_CREASE_TERMINAL_HIGH_MIN_FRAC`` of the profile) and this run
|
|
has not yet produced a comparable amount of high-power time at or after the
|
|
same position, the spin is still ahead.
|
|
|
|
Deliberately compares EVENTS, not clock positions: mapping the profile's
|
|
last high sample onto elapsed time and clearing there delays the reported
|
|
finalise by 16 s and then splits the wash anyway, because a run's spin can
|
|
arrive hundreds of seconds later than the profile's (the same
|
|
load-dependent duration spread behind #393).
|
|
|
|
Delay-only and bounded: never blocks past
|
|
``ANTI_CREASE_SPIN_WAIT_MAX_RATIO`` x expected, and fails open on any
|
|
missing input, so it cannot reproduce the #296 hang. The cap reads the
|
|
gates' programme time (item 511).
|
|
"""
|
|
block = self._matched_terminal_high
|
|
if block is None:
|
|
return False
|
|
start_frac, block_seconds = block[0], block[1]
|
|
if start_frac < ANTI_CREASE_TERMINAL_HIGH_MIN_FRAC:
|
|
return False # the profile's tail is genuinely low-power (#296 shape)
|
|
expected = self._expected_duration
|
|
start = self._current_cycle_start
|
|
if expected <= 0 or start is None:
|
|
return False
|
|
current_duration = self._gate_elapsed_s(timestamp)
|
|
if current_duration >= expected * ANTI_CREASE_SPIN_WAIT_MAX_RATIO:
|
|
return False # cap: waited long enough, let the finalise through
|
|
needed = block_seconds * ANTI_CREASE_TERMINAL_MATCH_FRAC
|
|
if needed <= 0:
|
|
return False
|
|
# Register item 196: scan from the block's ABSOLUTE offset on the profile's
|
|
# own grid when the store supplied one (element 3). `start_frac` is measured
|
|
# against the quiet-TRIMMED span - it has to be, or a capture's idle tail
|
|
# disarms the terminal gate above - while `expected` is the profile's
|
|
# avg_duration, which tracks the UNTRIMMED span. Their product is therefore a
|
|
# systematically LATE offset, and a late offset means the run's own spin sits
|
|
# BEFORE the scan window and is never counted, so the hold ran out the
|
|
# ANTI_CREASE_SPIN_WAIT_MAX_RATIO cap instead of releasing on the event. Over
|
|
# 36 real armed profile/cycle pairs the product recognised the spin 3 times
|
|
# and the absolute offset 14, with zero cases in either where the credited
|
|
# seconds exceeded the run's own terminal block (so no new premature-release
|
|
# exposure). The fallback keeps a pre-196 payload - an old state snapshot, the
|
|
# Playground, older callers - behaving exactly as before.
|
|
offset_s = float(block[2]) if len(block) >= 3 else start_frac * expected
|
|
# Element 4, when the store sent one, is the watts the block was measured
|
|
# against. Count the live seconds above the SAME bar or the two halves
|
|
# describe different things (register item 351).
|
|
ceiling_w = float(block[3]) if len(block) >= 4 else None
|
|
seen = self._high_power_seconds_since(offset_s, ceiling_w=ceiling_w)
|
|
if seen >= needed:
|
|
return False
|
|
if not self._anticrease_spin_wait_logged:
|
|
self._anticrease_spin_wait_logged = True
|
|
self._logger.debug(
|
|
"Finalize held, terminal high-power block still owed: '%s' ends "
|
|
"with a %.0fs block above %.0fW at %.0f%% of its run (scanning "
|
|
"from %.0fs); this cycle has %.0fs of it so far (elapsed %.0fs of "
|
|
"%.0fs expected).",
|
|
self._matched_profile,
|
|
block_seconds,
|
|
(
|
|
float(self._config.anti_wrinkle_max_power)
|
|
if ceiling_w is None
|
|
else ceiling_w
|
|
),
|
|
start_frac * 100.0,
|
|
offset_s,
|
|
seen,
|
|
current_duration,
|
|
expected,
|
|
)
|
|
return True
|
|
|
|
def _high_power_seconds_since(
|
|
self, offset_s: float, ceiling_w: float | None = None
|
|
) -> float:
|
|
"""Seconds this cycle has spent above ``ceiling_w`` at or after ``offset_s``
|
|
from its start (#399); ``anti_wrinkle_max_power`` when None.
|
|
|
|
The caller supplies the ceiling so both halves of the comparison use one
|
|
bar: the profile's block was measured against it too. The standby-band
|
|
path passes a share of the cycle's own peak (register item 351), the
|
|
anti-crease path passes nothing and keeps the dryer's tumble level.
|
|
|
|
Walks the readings backwards and stops at the offset, so the scan is bounded
|
|
by the tail of the trace rather than its whole length. Each reading covers
|
|
the interval up to the following one, which matches how the profile's own
|
|
block length is measured.
|
|
|
|
Two corrections to that per-interval credit, both of which decide whether the
|
|
guard releases:
|
|
|
|
* An outage-sized interval is unobserved time, not high-power time. Counting
|
|
it in full let a silent plug bank minutes of "spin" it never reported,
|
|
satisfy ``seen >= needed`` and release the finalise before the real
|
|
terminal spin - the #399 failure, reached by a different route. Same
|
|
treatment (and the same p95-derived ceiling) the tail scan at
|
|
``_smart_term_tail_stats`` and the gap-free quiet tally already apply.
|
|
Deliberately NOT ``_outage_threshold_s()``, which rebuilds a NumPy array
|
|
from every reading; this runs on the per-reading anti-crease path.
|
|
* When ``offset_s`` falls inside an interval, only the part after the offset
|
|
counts. Breaking out of the loop dropped that remainder entirely, and the
|
|
offset is ``start_frac * expected``, so a boundary reading is the norm
|
|
rather than an edge case.
|
|
"""
|
|
start = self._current_cycle_start
|
|
if start is None or not self._power_readings:
|
|
return 0.0
|
|
if self._stall_spans or self._user_pause_spans:
|
|
# The offset is on the profile's grid: programme time (items 511, 514).
|
|
offset_s = self._wall_offset_s(offset_s)
|
|
ceiling = (
|
|
float(self._config.anti_wrinkle_max_power)
|
|
if ceiling_w is None
|
|
else float(ceiling_w)
|
|
)
|
|
max_gap = min(3600.0, max(60.0, 10.0 * self._prior_p95_dt))
|
|
total = 0.0
|
|
readings = self._power_readings
|
|
for i in range(len(readings) - 1, -1, -1):
|
|
ts, power = readings[i]
|
|
elapsed = (ts - start).total_seconds()
|
|
if i + 1 >= len(readings):
|
|
continue # last reading covers no interval yet
|
|
next_elapsed = (readings[i + 1][0] - start).total_seconds()
|
|
if next_elapsed <= offset_s:
|
|
break # this interval ends at or before the offset, as do all earlier ones
|
|
interval = next_elapsed - elapsed
|
|
if interval > max_gap:
|
|
if elapsed < offset_s:
|
|
break
|
|
continue # unobserved time, not evidence of anything
|
|
if float(power) > ceiling:
|
|
# Credit only the portion at or after the offset.
|
|
total += next_elapsed - max(elapsed, offset_s)
|
|
return total
|
|
|
|
def _maybe_finalize_anticrease_tail(self, timestamp: datetime) -> bool:
|
|
"""Finalise a cycle that has entered the anti-crease tail into
|
|
STATE_ANTI_WRINKLE (#296). Returns True if the cycle was finalised.
|
|
|
|
Shared by the RUNNING / PAUSED / ENDING branches so the finalise fires no
|
|
matter which state a burst left the detector in. Uses Smart Termination
|
|
(in ``ANTI_WRINKLE_ELIGIBLE_REASONS``) so ``_finish_cycle`` routes into
|
|
STATE_ANTI_WRINKLE, which then absorbs the tail and splits off any next
|
|
wash on its first heating burst above ``anti_wrinkle_max_power``.
|
|
"""
|
|
if not self._is_anticrease_tail(timestamp):
|
|
return False
|
|
start_time = self._current_cycle_start or timestamp
|
|
current_duration = (timestamp - start_time).total_seconds()
|
|
# Register item 393a: the tail's baseline, from the window that was just
|
|
# accepted as tail (read before _finish_cycle clears the readings). Twice
|
|
# its lowest reading, never under the anti-wrinkle exit level.
|
|
tail_floor = max(
|
|
float(self._config.anti_wrinkle_exit_power),
|
|
float(self._config.stop_threshold_w),
|
|
2.0 * min(
|
|
(
|
|
float(p)
|
|
for ts, p in self._power_readings
|
|
if (timestamp - ts).total_seconds() <= ANTI_CREASE_CONFIRM_WINDOW_S
|
|
),
|
|
default=0.0,
|
|
),
|
|
)
|
|
self._logger.info(
|
|
"Anti-crease finalize: matched '%s' past expected %.0fs (elapsed %.0fs), "
|
|
"settled into the low-power tumble tail — finalizing into anti-wrinkle.",
|
|
self._matched_profile,
|
|
self._expected_duration,
|
|
current_duration,
|
|
)
|
|
self._finish_cycle(
|
|
timestamp,
|
|
status="completed",
|
|
termination_reason=TerminationReason.SMART,
|
|
keep_tail=True,
|
|
# Capped like the three sibling keep_tail finishes (#424). This path can
|
|
# fire on a window of *watchdog* keepalives: a change-only plug that has
|
|
# gone silent emits nothing, the injected 0 W readings satisfy the "all
|
|
# at or below anti_wrinkle_max_power" window, and `timestamp` is then
|
|
# the moment the watchdog noticed rather than the moment the appliance
|
|
# stopped - post-cycle standby banked as cycle time, which feeds
|
|
# avg_duration and self-amplifies.
|
|
#
|
|
# The tumble tail itself is never cut into, which is why this is safe
|
|
# here: an anti-crease baseline sits ABOVE stop_threshold
|
|
# (const.py:811-812, a ~2.5-3.2 W draw against a ~1.2 W threshold), so
|
|
# every one of those readings refreshes `_last_active_time`, and for a
|
|
# non-dishwasher `_keep_tail_cap` returns exactly that - the last
|
|
# tumble. A real tail therefore loses only the trailing quiet gap
|
|
# between its last reading and this finalize, which is time the
|
|
# appliance drew nothing, and a tail of genuinely-0 W readings is
|
|
# dropped at `_last_active_time`.
|
|
#
|
|
# NB: the cap is no longer `max(expected_end, _last_active_time)` and
|
|
# so is NOT bounded below by the expected end - on a washer or dryer
|
|
# the stored end can now land earlier than the matched profile's
|
|
# expected duration. Shorten-only still holds; "never earlier than
|
|
# expected_end" no longer does, and this paragraph used to claim it.
|
|
tail_cap=self._keep_tail_cap(start_time),
|
|
)
|
|
# After _finish_cycle, whose reset() clears it (register item 393a).
|
|
if self._state == STATE_ANTI_WRINKLE:
|
|
self._anticrease_tail_floor_w = tail_floor
|
|
return True
|
|
|
|
def _is_terminal_drop(self) -> bool:
|
|
"""Whether the current low-power event is an anomalously-early hard drop.
|
|
|
|
Mirrors ``_ml_end_confidence``: builds the offset-second trace from the
|
|
current cycle's readings and asks the injected terminal-drop provider.
|
|
Returns ``False`` when there is no provider, no cycle start, or the
|
|
provider declines/raises (ML off / too little history / not anomalous),
|
|
so the caller keeps the proven soak-bridging end-detection.
|
|
"""
|
|
provider = self._terminal_drop_provider
|
|
start = self._current_cycle_start
|
|
if provider is None or start is None or not self._power_readings:
|
|
return False
|
|
# Throttle: reuse within the window, scoped to this cycle + expected_duration.
|
|
now_ts = self._power_readings[-1][0]
|
|
exp = float(self._expected_duration)
|
|
cache = self._terminal_drop_cache
|
|
if (
|
|
cache is not None
|
|
and cache[1] == exp
|
|
and cache[2] == start
|
|
and (now_ts - cache[0]).total_seconds() < ML_PROVIDER_THROTTLE_SECONDS
|
|
):
|
|
return cache[3]
|
|
points = [
|
|
((ts - start).total_seconds(), float(power))
|
|
for ts, power in self._power_readings
|
|
]
|
|
try:
|
|
result = bool(provider(points, exp))
|
|
except Exception: # noqa: BLE001 - ML must never break detection
|
|
result = False
|
|
self._terminal_drop_cache = (now_ts, exp, start, result)
|
|
return result
|
|
|
|
def _should_defer_finish(self, duration: float) -> bool:
|
|
"""Check if we should defer termination based on expected duration."""
|
|
# Check explicit verified pause override from manager
|
|
if getattr(self, "_verified_pause", False):
|
|
self._logger.debug("Deferring cycle finish: Verified pause active")
|
|
return True
|
|
|
|
# Dishwasher minimum-duration floor: even without a matched profile (e.g.
|
|
# first cycle of a program, or the 5-min matcher hasn't fired yet) a
|
|
# dishwasher cycle should never end before it has crossed the minimum
|
|
# reasonable programme duration. This prevents a dip during the fill or
|
|
# early wash phase from being read as the end of a complete cycle.
|
|
#
|
|
# A MATCHED profile overrides the blanket constant with its own learned
|
|
# length, because the constant is a stand-in for exactly the knowledge a
|
|
# match supplies - as this comment's own "even without a matched profile"
|
|
# says. Blanket, it is wrong for real hardware: the community catalogue
|
|
# carries a 6.0 min Smeg "Delay- prewash", which a 30 min floor defers by
|
|
# half an hour. The floor still applies unmatched, and a matched profile
|
|
# can only ever LOWER it (`min`), never license a longer deferral - the
|
|
# 39.4 min Electrolux "Rapido" already clears it and is unaffected.
|
|
#
|
|
# Gated on a TRUSTED match, not merely a present one. This floor is an
|
|
# anti-premature-end guard, so the risk is the opposite way round from
|
|
# the ENDING fallback gate (item 329, where a confidence check measured
|
|
# as pure cost): getting this wrong ends a dishwasher during its fill or
|
|
# early-wash dip and records the rest of the programme as a second
|
|
# cycle, which is the expensive failure. A low-confidence match to a
|
|
# short look-alike is exactly how that happens, so it does not get to
|
|
# lower the bar; nor does an ambiguous one. (It also honoured the #364
|
|
# prefix-fit flag until that was removed in 0.5.8.) No-op on the whole
|
|
# corpus either way - every corpus dishwasher profile is over 90 minutes.
|
|
_dw_floor = DISHWASHER_MIN_CYCLE_DURATION_S
|
|
if (
|
|
self._matched_profile
|
|
and self._expected_duration > 0
|
|
and self._last_match_confidence >= self._config.match_confidence_threshold
|
|
and not self._match_ambiguous
|
|
):
|
|
_dw_floor = min(_dw_floor, float(self._expected_duration))
|
|
if self._config.device_type == "dishwasher" and duration < _dw_floor:
|
|
self._logger.debug(
|
|
"Deferring dishwasher cycle end: elapsed %.0fs < minimum %.0fs",
|
|
duration,
|
|
_dw_floor,
|
|
)
|
|
return True
|
|
|
|
if not self._matched_profile or self._expected_duration <= 0:
|
|
return False
|
|
|
|
# Safety: Don't defer forever
|
|
if duration > (self._expected_duration + DEFAULT_MAX_DEFERRAL_SECONDS):
|
|
self._logger.warning(
|
|
"Deferral limit exceeded (%.0fs > expected %.0f + %s), allowing finish",
|
|
duration,
|
|
self._expected_duration,
|
|
DEFAULT_MAX_DEFERRAL_SECONDS,
|
|
)
|
|
return False
|
|
|
|
# Opt-in ML end-guard (asymmetric anti-premature-stop, bounded). If the
|
|
# cycle-end model judges this low-power event to be more likely a pause
|
|
# than the true end, defer the normal completion - but only for a bounded
|
|
# extra window, so a wrong model can delay, never hang, the cycle. As the
|
|
# low-power run lengthens the model's confidence rises, so a genuine end
|
|
# is released once the model agrees or the cap is reached.
|
|
if (
|
|
self._end_confidence_provider is not None
|
|
and self._last_match_confidence >= DEFAULT_DEFER_FINISH_CONFIDENCE
|
|
):
|
|
confidence = self._ml_end_confidence()
|
|
if confidence is not None and confidence < ML_END_GUARD_MIN_CONFIDENCE:
|
|
if self._ml_defer_start_duration is None:
|
|
self._ml_defer_start_duration = duration
|
|
if (duration - self._ml_defer_start_duration) < ML_END_GUARD_MAX_DEFER_SECONDS:
|
|
self._logger.debug(
|
|
"Deferring cycle finish: ML end-guard (P(true end)=%.2f < %.2f)",
|
|
confidence,
|
|
ML_END_GUARD_MIN_CONFIDENCE,
|
|
)
|
|
return True
|
|
elif confidence is not None:
|
|
# Model is confident this is the true end -> stop ML-deferring.
|
|
self._ml_defer_start_duration = None
|
|
|
|
# Dishwasher passive drying protection:
|
|
# Dishwashers can have 2+ hour passive drying phases at near-0W. A terminal
|
|
# drain spike that fires early in the ENDING state (e.g. at 120 min of a
|
|
# 233-min ECO cycle) resets _time_below_threshold, and the subsequent 60-min
|
|
# silence timeout would otherwise end the cycle at ~180 min - well before the
|
|
# real finish. Defer until the cycle reaches the late-phase threshold (the
|
|
# same one used by the end-spike arm gate, so both move together) so that
|
|
# smart termination can catch the true end (~99% of expected) instead.
|
|
# Confidence may be low this early, so the normal confidence gate is
|
|
# bypassed here.
|
|
if (
|
|
self._config.device_type == "dishwasher"
|
|
and self._matched_profile
|
|
and self._expected_duration > 0
|
|
and duration
|
|
< (self._expected_duration * DISHWASHER_END_SPIKE_MIN_PROGRESS)
|
|
):
|
|
self._logger.debug(
|
|
"Deferring cycle finish: dishwasher drying phase protection "
|
|
"(%.0fs < %.0f%% of expected %.0fs, profile: %s, conf %.2f)",
|
|
duration,
|
|
DISHWASHER_END_SPIKE_MIN_PROGRESS * 100,
|
|
self._expected_duration,
|
|
self._matched_profile,
|
|
self._last_match_confidence,
|
|
)
|
|
return True
|
|
|
|
# Issue #43: dishwasher end-spike wait protection. Once past the 85%
|
|
# passive-drying gate above, we still keep the cycle deferred until
|
|
# the real end-of-cycle pump-out fires (sets _end_spike_seen=True via
|
|
# the 85% progress gate in STATE_ENDING) or we cross the
|
|
# smart-termination wait window (expected + 30 min) - whichever comes
|
|
# first. Shares DISHWASHER_END_SPIKE_WAIT_SECONDS with Smart
|
|
# Termination's wait branch so the two paths release the cycle at the
|
|
# same instant. Beyond the wait window, Smart Termination's
|
|
# past_wait_period kicks in and finalises; below it, the fallback
|
|
# timeout's energy gate is the safety net for cycles whose pump-out
|
|
# never arrives.
|
|
# Mirrors the STATE_ENDING pump-out wait so both paths release together.
|
|
# Keep deferring while we are still inside the wait window, UNLESS the cycle
|
|
# has already reached its expected duration and has since been sustained-quiet
|
|
# for DISHWASHER_END_SPIKE_QUIET_RELEASE_SECONDS - in which case any terminal
|
|
# pump-out has already happened, so a cycle that finished slightly short of the
|
|
# profile's (drifted-up) average is released here instead of hanging to
|
|
# expected + 30 min. The ``duration >= expected`` gate keeps a long
|
|
# passive-drying phase that still precedes a late pump-out deferred.
|
|
quiet_released = (
|
|
duration >= self._expected_duration
|
|
and self._time_below_threshold_gapfree
|
|
>= self._dishwasher_quiet_release_s()
|
|
)
|
|
if (
|
|
self._config.device_type == "dishwasher"
|
|
and self._matched_profile
|
|
and self._expected_duration > 0
|
|
and not self._end_spike_seen
|
|
and duration
|
|
< (self._expected_duration + self._dishwasher_end_spike_wait_s())
|
|
and not quiet_released
|
|
):
|
|
# Report the gap-free tally: that is what `quiet_released` above reads,
|
|
# and after a telemetry outage the two diverge - logging the plain one
|
|
# would show quiet time that played no part in the decision.
|
|
self._logger.debug(
|
|
"Deferring cycle finish: dishwasher waiting for end-of-cycle "
|
|
"pump-out (%.0fs < expected %.0fs + %.0fs wait, observed quiet "
|
|
"%.0fs of %.0fs needed, profile: %s)",
|
|
duration,
|
|
self._expected_duration,
|
|
self._dishwasher_end_spike_wait_s(),
|
|
self._time_below_threshold_gapfree,
|
|
self._dishwasher_quiet_release_s(),
|
|
self._matched_profile,
|
|
)
|
|
return True
|
|
|
|
# If matched profile, enforce min duration ratio
|
|
ratio = self._config.min_duration_ratio
|
|
|
|
# --- STRICTER DEFERRAL ---
|
|
# If we are NOT in a verified pause, but power has been low for a long time (ENDING state),
|
|
# we only defer if we are VERY confident this profile is correct.
|
|
# This prevents hanging on too-long profiles that matched early but are now diverging.
|
|
if self._last_match_confidence < DEFAULT_DEFER_FINISH_CONFIDENCE:
|
|
self._logger.debug(
|
|
"Not deferring finish: confidence %.2f too low for unverified pause (profile: %s)",
|
|
self._last_match_confidence,
|
|
self._matched_profile,
|
|
)
|
|
return False
|
|
|
|
# Primary check: Is duration significantly below expectation?
|
|
if duration < (self._expected_duration * ratio):
|
|
self._logger.debug(
|
|
"Deferring cycle finish: duration %.0fs < %.0f%% of expected %.0fs (profile: %s, confidence %.2f)",
|
|
duration,
|
|
ratio * 100,
|
|
self._expected_duration,
|
|
self._matched_profile,
|
|
self._last_match_confidence,
|
|
)
|
|
return True
|
|
|
|
# (A "ratio to 1 + profile_duration_tolerance" window used to follow, but both
|
|
# of its branches returned False - the tolerance changed nothing here.)
|
|
return False
|
|
|
|
def _fallback_shortening_bar(
|
|
self, expected: float, ambiguous: bool, longest: float
|
|
) -> float | None:
|
|
"""Elapsed seconds from which the ENDING fallback shortens, None when blocked.
|
|
|
|
The item 306/330/355 bar for a match with these values: ``expected`` at the
|
|
device's late ratio, or, while the match is ambiguous, the longest candidate
|
|
at the plain ``END_GATE_LATE_RATIO`` when that is longer, and blocked when
|
|
there are no candidate durations to compare (see the gate in
|
|
``process_reading``). Shared by that gate and
|
|
:meth:`ambiguous_ending_match_defers`, so the two cannot disagree.
|
|
"""
|
|
bar = float(expected)
|
|
raised = False
|
|
if ambiguous:
|
|
if longest > bar:
|
|
bar = float(longest)
|
|
raised = True
|
|
elif longest <= 0.0:
|
|
return None
|
|
ratio = (
|
|
END_GATE_LATE_RATIO
|
|
if raised
|
|
else resolve_end_gate_late_ratio(self._config.device_type)
|
|
)
|
|
return ratio * bar
|
|
|
|
def _ending_wait_left_s(
|
|
self, expected: float, ambiguous: bool, longest: float, confidence: float, cat: Any
|
|
) -> float:
|
|
"""Seconds until the ENDING fallback would finish under continued quiet, for
|
|
a match with these values (register item 469b).
|
|
|
|
The fallback gate in ``process_reading``, run forward: the hazard wait
|
|
before the shortening bar, the shortened wait after it (the gate swaps one
|
|
for the other, so the shortening can also lengthen), then
|
|
``_should_defer_finish``'s duration floor, which needs the confidence. The
|
|
energy gate and Smart Termination are left out: neither depends on which
|
|
programme is matched in a way an ambiguous tick changes, except that the
|
|
ambiguity blocks Smart Termination (see the caller).
|
|
"""
|
|
now = self._power_readings[-1][0]
|
|
elapsed = self._gate_elapsed_s(now) # the gate's own clock (item 511)
|
|
quiet = float(self._time_below_threshold)
|
|
hazard = self._hazard_wait_for(
|
|
now, max(self._config.off_delay, self._config.min_off_gap), cat, expected, ambiguous
|
|
)
|
|
left = max(0.0, hazard - quiet)
|
|
bar = self._fallback_shortening_bar(expected, ambiguous, longest)
|
|
if bar is not None and elapsed + left >= bar:
|
|
short = max(
|
|
self._config.off_delay,
|
|
min(self._config.min_off_gap, END_GATE_LATE_SECONDS),
|
|
)
|
|
left = max(0.0, short - quiet, bar - elapsed)
|
|
if confidence >= DEFAULT_DEFER_FINISH_CONFIDENCE:
|
|
left = max(left, expected * float(self._config.min_duration_ratio) - elapsed)
|
|
return left
|
|
|
|
def ambiguous_ending_match_defers(
|
|
self, expected: Any, confidence: Any, longest: Any, catalogue: Any = None
|
|
) -> bool:
|
|
"""Would an AMBIGUOUS match with these values make this ENDING wait longer?
|
|
|
|
Register item 469(b). A tick in ENDING matches a trace that ends in its idle
|
|
tail, which reads as the prefix of a longer programme pausing, so its top-1
|
|
drifts longer and ties with its runner-up. Applied as is, a shorter match
|
|
interval (what "Apply all" suggests) let such a tick defer the end: 01KXGA3C
|
|
62f39dfc34f4, a 29 min wool wash, waited 23.2 min instead of 6.7 min at an
|
|
87 s interval. ``update_match`` refuses such a match when this says the
|
|
fallback would finish later with it (:meth:`_ending_wait_left_s` for both),
|
|
or as late when the current match is clear (only the ambiguous one blocks
|
|
Smart Termination). Not "a longer programme": the shortening bar is not
|
|
monotone in the expected duration. A match to the LONGEST candidate does
|
|
not raise it, so it can end sooner than a shorter ambiguous one whose bar
|
|
the longest raised (01KBWSV8 1dbc19ccba79 / daea1437efcc: 6.7 min applied,
|
|
23-25 min held), and a shorter one can raise it (01KXGA3C 9c2624675652).
|
|
A dishwasher's drying and pump-out waits follow the expected duration
|
|
alone, so there a longer one is refused outright.
|
|
|
|
False outside ENDING or without a matched programme: nothing to keep.
|
|
"""
|
|
if (
|
|
self._state != STATE_ENDING
|
|
or not self._matched_profile
|
|
or self._expected_duration <= 0
|
|
or self._current_cycle_start is None
|
|
or not self._power_readings
|
|
):
|
|
return False
|
|
try:
|
|
new_expected = float(expected)
|
|
new_confidence = float(confidence or 0.0)
|
|
new_longest = float(longest or 0.0)
|
|
except (TypeError, ValueError, OverflowError):
|
|
return True
|
|
if not (math.isfinite(new_expected) and new_expected > 0):
|
|
return True # would unmatch the cycle: the unshortened fallback
|
|
if (
|
|
self._config.device_type == DEVICE_TYPE_DISHWASHER
|
|
and new_expected > self._expected_duration
|
|
):
|
|
return True
|
|
new_left = self._ending_wait_left_s(
|
|
new_expected, True, new_longest, new_confidence,
|
|
self._sanitize_pause_catalogue(catalogue),
|
|
)
|
|
old_left = self._ending_wait_left_s(
|
|
self._expected_duration, self._match_ambiguous,
|
|
self._longest_candidate_duration, self._last_match_confidence,
|
|
self._matched_pause_catalogue,
|
|
)
|
|
if new_left > old_left + 1.0:
|
|
return True
|
|
return not self._match_ambiguous and new_left >= old_left - 1.0
|
|
|
|
def _dishwasher_quiet_release_s(self) -> float:
|
|
"""Sustained quiet past expected that releases the pump-out wait (#379).
|
|
|
|
The configured value - raised to DISHWASHER_QUIET_RELEASE_TERMINAL_MARGIN x
|
|
the matched profile's measured quiet before its terminal event while this
|
|
run has NOT yet been through that quiet (register item 392). The release
|
|
exists for a pump-out that already happened or never comes; a run still
|
|
short of the quiet its programme always has before the pump-out has not
|
|
reached that point. A run that has been through it - #424's Beko and the
|
|
Hatton ECO, whose element-11 event sits mid-cycle - keeps the configured
|
|
value (or the item-465 floor below). Lengthen-only; the 30 min spike wait
|
|
still bounds the whole wait.
|
|
|
|
Also never shorter than the same margin x the longest below-stop pause
|
|
the profile's traced evidence ever came back from (element 14, register
|
|
item 465), whether or not the run has been through its terminal quiet.
|
|
Element 11 is a MEDIAN over the profile's last events, and a cycle closed
|
|
before its pump-out contributes the wrong event: on 01KGM619's Eco half
|
|
the stored cycles end on their last heating block (100-160 s gap), the
|
|
rest on a pump-out 4840-4860 s after it, so the median lands at 2500 s,
|
|
a quiet no cycle has. 1.1 x that released 6 of 6 pump-out cycles 6-11 min
|
|
before their pump-out, and each stored ~8.1k s instead of ~11.1k s, which
|
|
drags `avg_duration`, and with it the next release, earlier. A pause the
|
|
programme has resumed from says "activity can still follow a quiet this
|
|
long" directly. No position filter: the release is only asked past the
|
|
expected end, and the catalogue's fractions are of each cycle's own span,
|
|
which is longer than `expected` on exactly the cycles that kept their
|
|
pump-out. Lengthen-only, bounded by the same spike wait.
|
|
"""
|
|
base = max(
|
|
float(self._config.dishwasher_end_spike_quiet_release),
|
|
self._resumed_pause_release_s(),
|
|
)
|
|
quiet = self._matched_terminal_quiet_s
|
|
if not (self._matched_profile and quiet) or not self._power_readings:
|
|
return base
|
|
now = self._power_readings[-1][0]
|
|
if self._spike_follows_terminal_quiet(now, latest=True):
|
|
return base
|
|
return max(base, DISHWASHER_QUIET_RELEASE_TERMINAL_MARGIN * float(quiet))
|
|
|
|
def _resumed_pause_release_s(self) -> float:
|
|
"""Margin x the longest pause the matched profile resumed from, else 0.
|
|
|
|
Read from the hazard gate's catalogue (element 14) with its evidence bar
|
|
(END_GATE_HAZARD_MIN_CYCLES traced cycles). See
|
|
:meth:`_dishwasher_quiet_release_s` (register item 465).
|
|
"""
|
|
cat = self._matched_pause_catalogue
|
|
if (
|
|
cat is None
|
|
or cat[0] < END_GATE_HAZARD_MIN_CYCLES
|
|
or not cat[1]
|
|
or not self._matched_profile
|
|
):
|
|
return 0.0
|
|
return DISHWASHER_QUIET_RELEASE_TERMINAL_MARGIN * max(d for _f, d in cat[1])
|
|
|
|
def _spike_follows_terminal_quiet(
|
|
self, timestamp: datetime, *, latest: bool = False
|
|
) -> bool:
|
|
"""May this ENDING spike be the dishwasher's terminal pump-out? (item 392)
|
|
|
|
True unless the matched profile's quiet before its terminal event has been
|
|
measured (element 11) and this run has not yet been through it - asked
|
|
with the same `terminal_quiet_seen` the keep-tail cap and the banked-tail
|
|
repair use. On the corpus's "65° full" (934 s measured, 17/17 cycles) a
|
|
24 W fan blip 211 s after the last heating armed the end at 85% of
|
|
expected, and Smart Termination closed the cycle 12 min before the real
|
|
pump-out. Only ever withholds the arm, so it can delay an end, never
|
|
bring one forward; every other device type, and an unmeasured profile,
|
|
is unaffected.
|
|
"""
|
|
quiet = self._matched_terminal_quiet_s
|
|
start = self._current_cycle_start
|
|
if (
|
|
self._config.device_type != DEVICE_TYPE_DISHWASHER
|
|
or not self._matched_profile
|
|
or not quiet
|
|
or start is None
|
|
or not self._power_readings
|
|
):
|
|
return True
|
|
# Memoised per reading: the release asks this on every ENDING reading past
|
|
# the expected end, and the answer only changes when a reading arrives.
|
|
key = (len(self._power_readings), timestamp, latest)
|
|
hit = self._terminal_quiet_memo
|
|
if hit is not None and hit[0] == key:
|
|
return hit[1]
|
|
pts = [((ts - start).total_seconds(), float(pw)) for ts, pw in self._power_readings]
|
|
last_off = (timestamp - start).total_seconds()
|
|
if latest:
|
|
# "Has the run been through it by now?": measured from the last
|
|
# activity, not from a spike that has not happened.
|
|
last_active = self._last_active_time
|
|
if last_active is None:
|
|
return True
|
|
last_off = (last_active - start).total_seconds()
|
|
seen = terminal_quiet_seen(
|
|
pts,
|
|
last_off,
|
|
self._config.stop_threshold_w,
|
|
float(quiet),
|
|
TERMINAL_EVENT_PEAK_FRAC,
|
|
)
|
|
self._terminal_quiet_memo = (key, seen)
|
|
return seen
|
|
|
|
def _dishwasher_end_spike_wait_s(self) -> float:
|
|
"""Grace past the expected end while waiting for the terminal pump-out.
|
|
|
|
Capped at the programme's OWN expected length - the same ``min()`` shape
|
|
register item 331 gave ``DISHWASHER_MIN_CYCLE_DURATION_S``, for the same
|
|
reason. A flat 1800 s is 20% of a 150 min ECO cycle but **five times** a
|
|
6 min Smeg "Delay- prewash", and the community catalogue carries exactly
|
|
that programme. Asymmetric: the cap can only ever SHORTEN the wait, never
|
|
extend it, so no cycle waits longer than it does today.
|
|
|
|
A no-op across the maintainer's corpus, where every dishwasher profile is
|
|
>90 min and the cap therefore never binds (register item 357). Unmatched
|
|
cycles keep the flat constant: with no expected duration there is nothing
|
|
to be proportional to.
|
|
"""
|
|
expected = float(self._expected_duration or 0.0)
|
|
if expected <= 0:
|
|
return DISHWASHER_END_SPIKE_WAIT_SECONDS
|
|
return min(DISHWASHER_END_SPIKE_WAIT_SECONDS, expected)
|
|
|
|
def _ending_hard_finalize_quiet_s(self) -> float:
|
|
"""Continuous sub-threshold span the ENDING backstop requires.
|
|
|
|
Same cap, same reason: 600 s of required quiet is a third of a 30 min
|
|
programme and longer than a 6 min one, which would disarm the backstop
|
|
entirely on a short programme - the opposite of what a safety net is for.
|
|
The ``off_delay`` / ``min_off_gap`` floor is applied by the caller and is
|
|
unaffected.
|
|
"""
|
|
expected = float(self._expected_duration or 0.0)
|
|
if expected <= 0:
|
|
return ENDING_HARD_FINALIZE_MIN_QUIET_S
|
|
return min(ENDING_HARD_FINALIZE_MIN_QUIET_S, expected)
|
|
|
|
def _keep_tail_cap(self, start_time: datetime) -> datetime | None:
|
|
"""Latest end time a *kept* tail may claim (#424).
|
|
|
|
The paths that keep their tail do so because the tail can be real cycle
|
|
time: a dishwasher's near-0 W passive drying phase sits between the last
|
|
drain spike and the actual end of the programme (issue #43), so snapping
|
|
back to ``_last_active_time`` would store a falsely short cycle. But they
|
|
fire on accumulated quiet time, and on a publish-on-change plug that wait
|
|
is minutes of *post-appliance* standby - which stamping ``timestamp`` as
|
|
the end time banked as cycle time.
|
|
|
|
Measured on the #424 reporter's dishwasher: every cycle that ended via
|
|
`timeout` stored a 0-29 s tail (237-239 min, matching the appliance), and
|
|
every cycle that ended via `smart` stored a 96-1239 s tail (240-260 min),
|
|
with ``_last_active_time`` unchanged across the whole history - only the
|
|
termination path differed. It also self-amplifies, because the inflated
|
|
duration feeds ``avg_duration``, which raises ``expected_duration``,
|
|
which delays the next Smart Termination further (the second reporter's
|
|
profile had already drifted 63 -> 70.5 min).
|
|
|
|
The cap used to be the matched profile's **expected end**. That is the
|
|
mean of these same stored durations, so it moved with the thing it was
|
|
bounding: a banked tail raised ``avg_duration``, the higher average
|
|
allowed a longer tail, and the reported end drifted later every run.
|
|
Measured over 375 cycles from 16 devices, smart-terminated cycles banked a
|
|
median **12.6 min** of post-appliance time (washing machines **22.7 min**,
|
|
p90 40.4 min) against ~0 min for every other termination path, and the
|
|
profiles carried a mean **+5.3%** duration inflation as a result - on the
|
|
#427 reporter's washer, +20.2 min on a 108 min programme, which is also
|
|
why their ETA read 131 min for a ~105 min wash.
|
|
|
|
So anchor on the last real activity instead. Where a passive phase can
|
|
legitimately follow it, allow only what this programme has been
|
|
**measured** to do (``profile_terminal_quiet_seconds``, element 11) - a
|
|
statistic taken from the traces, not from the stored durations, so it
|
|
cannot be inflated by the tail it bounds, and gated on having been seen
|
|
repeatedly rather than once.
|
|
|
|
**Only a dishwasher has a passive terminal phase.** Every other type ends
|
|
on activity - a washer's spin, a dryer's drum - which ``_last_active_time``
|
|
already marks, so there is nothing legitimate to bank after it. That is
|
|
not a new assumption: the fallback-timeout path beside this one has always
|
|
read ``keep_tail = device_type == "dishwasher"``. Smart Termination was
|
|
the one path that kept a tail for every type, which is exactly where the
|
|
22.7 min washing-machine median came from.
|
|
|
|
Within the dishwasher case, two sub-cases, and the difference matters:
|
|
|
|
* the run produced its terminal pump-out (``_end_spike_seen``).
|
|
``_last_active_time`` already sits on it, so that IS the end.
|
|
* it did not, so the programme ended in its passive drying phase. Allow up
|
|
to the profile's measured quiet span past the last activity.
|
|
|
|
Falls back to the old expected-end cap when that span has not been
|
|
measured, rather than truncating a drying phase on no evidence - the
|
|
measured corpus shows the pump-out missing in a substantial minority of
|
|
runs on some machines, and in those runs the drying IS the tail. Still
|
|
asymmetric and shorten-only; still None for an unmatched cycle.
|
|
"""
|
|
if self._expected_duration <= 0:
|
|
return None
|
|
expected_end = start_time + timedelta(seconds=self._expected_duration)
|
|
last_active = self._last_active_time
|
|
if last_active is None:
|
|
return expected_end
|
|
if self._config.device_type != DEVICE_TYPE_DISHWASHER:
|
|
return last_active
|
|
cap = self._dishwasher_tail_cap(start_time, expected_end, last_active)
|
|
# Never before the length the user has vouched for (register item 384),
|
|
# the floor `ProfileStore.async_repair_banked_tails` applies, so live and
|
|
# the repair store the same duration. Element 11 is measured from the
|
|
# profile's own traces, and on a machine that dries silently AFTER its
|
|
# last activity it reads the pre-drying pause instead: the Hatton ECO
|
|
# export stores 230-235 min (one corrected to 234), and without this the
|
|
# cap snapped three of its four cycles to their last activity at ~120 min.
|
|
# Lengthen-only, and only for a run that has itself lasted that long: a
|
|
# cancelled or short run never reaches the floor, and lifting its cap
|
|
# would bank its whole post-appliance wait (#424). Same outcome as the
|
|
# repair, which never lengthens a cycle.
|
|
trusted = self._matched_trusted_min_s
|
|
if trusted and self._power_readings:
|
|
floor = start_time + timedelta(seconds=TRUSTED_LENGTH_FLOOR_FRAC * trusted)
|
|
if cap < floor <= self._power_readings[-1][0]:
|
|
return floor
|
|
return cap
|
|
|
|
def _dishwasher_tail_cap(
|
|
self, start_time: datetime, expected_end: datetime, last_active: datetime
|
|
) -> datetime:
|
|
"""The dishwasher half of :meth:`_keep_tail_cap`, before the trusted floor."""
|
|
# Only a spike LATE enough to be the terminal pump-out licenses snapping
|
|
# the stored end back to the last activity. `_end_spike_seen` is set from
|
|
# DISHWASHER_END_SPIKE_MIN_PROGRESS (0.85), but this file does not treat
|
|
# every such spike as terminal: `_resolve_smart_ratio` relaxes its gate
|
|
# only at `>= expected * 0.90`, because below that the spike can be the
|
|
# pre-final-rinse drain with a passive Dry phase still to come. Capping
|
|
# at `last_active` for an 87% drain cuts that drying off, which lowers
|
|
# `avg_duration`, which makes the NEXT Smart Termination fire earlier -
|
|
# the error compounds in the direction that splits cycles. Same 0.90
|
|
# test here, so a pre-rinse drain falls through to the measured quiet
|
|
# span or the expected-end fallback below.
|
|
if getattr(self, "_end_spike_seen", False) and getattr(
|
|
self, "_end_spike_duration", 0.0
|
|
) >= self._expected_duration * 0.90:
|
|
return last_active
|
|
quiet = self._matched_terminal_quiet_s
|
|
if quiet is None:
|
|
return max(expected_end, last_active)
|
|
# ...and if the drying ALREADY happened, adding the allowance on top
|
|
# counts the same quiet twice. The 0.90 test above only catches a spike
|
|
# late enough to be unambiguously terminal; a pump-out at 85-90% of
|
|
# expected falls through it, and on a machine that dries BEFORE its final
|
|
# drain that banks a second drying period into the stored duration, which
|
|
# feeds `avg_duration` - the exact drift item 297 exists to remove.
|
|
# `ProfileStore.async_repair_banked_tails` has always asked the trace this
|
|
# question; this path did not, so the same cycle got one duration live and
|
|
# another when the repair re-judged it. Same helper, same 0.5 bar, so the
|
|
# two cannot drift again (register item 347).
|
|
#
|
|
# And it has to ask at the threshold the allowance was MEASURED at (#424):
|
|
# see `terminal_quiet_seen` for the Beko that banked 10 min on every cycle
|
|
# because this test ran at the stop threshold while the statistic it
|
|
# guards is taken at a fraction of the cycle's peak.
|
|
if self._current_cycle_start is not None and self._power_readings:
|
|
_start = self._current_cycle_start
|
|
_pts = [
|
|
((ts - _start).total_seconds(), float(pw))
|
|
for ts, pw in self._power_readings
|
|
]
|
|
_last_off = (last_active - _start).total_seconds()
|
|
if terminal_quiet_seen(
|
|
_pts,
|
|
_last_off,
|
|
self._config.stop_threshold_w,
|
|
float(quiet),
|
|
TERMINAL_EVENT_PEAK_FRAC,
|
|
):
|
|
# ...at the terminal event, which may sit below the stop
|
|
# threshold and so after `last_active` (register item 384).
|
|
return _start + timedelta(
|
|
seconds=terminal_event_end(_pts, _last_off, TERMINAL_EVENT_PEAK_FRAC)
|
|
)
|
|
return last_active + timedelta(seconds=min(quiet, TERMINAL_QUIET_CAP_S))
|
|
|
|
def _finish_cycle(
|
|
self,
|
|
timestamp: datetime,
|
|
status: str = "completed",
|
|
termination_reason: str = TerminationReason.TIMEOUT,
|
|
keep_tail: bool = False,
|
|
tail_cap: datetime | None = None,
|
|
) -> None:
|
|
"""Finalize cycle.
|
|
|
|
Args:
|
|
timestamp: Time of completion
|
|
status: Cycle status string
|
|
termination_reason: Reason for termination
|
|
keep_tail: If True, use current timestamp as end time and preserve
|
|
trailing zero readings (e.g. Smart Termination).
|
|
If False (default), snap back to last active time and trim
|
|
trailing zeros (e.g. Timeout).
|
|
tail_cap: Latest end time a kept tail may claim. Ignored when
|
|
``keep_tail`` is False or when it is not earlier than
|
|
``timestamp``; readings past it are dropped so the stored
|
|
trace and the stored duration stay consistent.
|
|
"""
|
|
|
|
# Capture data before reset
|
|
readings = self._power_readings
|
|
if keep_tail:
|
|
end_time = timestamp
|
|
if tail_cap is not None and tail_cap < end_time:
|
|
end_time = tail_cap
|
|
readings = [r for r in readings if r[0] <= end_time]
|
|
else:
|
|
end_time = self._last_active_time or timestamp
|
|
|
|
if not self._current_cycle_start:
|
|
self.reset(timestamp=timestamp)
|
|
return
|
|
|
|
duration = (end_time - self._current_cycle_start).total_seconds()
|
|
|
|
# "Interrupted" logic (short cycle etc)
|
|
if duration < self._config.interrupted_min_seconds:
|
|
status = "interrupted"
|
|
elif duration < self._config.completion_min_seconds:
|
|
status = "interrupted"
|
|
|
|
# Trim leading/trailing zero readings for cleaner data
|
|
# If we keep tail, we explicitly do NOT trim end zeros
|
|
trimmed_readings = trim_zero_readings(
|
|
readings,
|
|
threshold=self._config.stop_threshold_w,
|
|
trim_end=not keep_tail,
|
|
)
|
|
|
|
# Ensure power_data covers the full duration until end_time
|
|
# (especially important for manual recordings or drying phases with no sensor updates)
|
|
final_readings = list(trimmed_readings)
|
|
if final_readings:
|
|
last_t, last_p = final_readings[-1]
|
|
if last_t < end_time:
|
|
# Not at ACTIVE power when a capped tail ends past the last kept
|
|
# sample: interpolation would read the whole drying allowance (up
|
|
# to TERMINAL_QUIET_CAP_S) as full draw, in the stored trace and in
|
|
# every envelope built from it. Readings past a cap at or after
|
|
# `_last_active_time` are quiet by construction, so the first of
|
|
# them is a real observation of this level - the rule the
|
|
# banked-tail repair applies to the same point (register item 384).
|
|
if last_p >= self._config.stop_threshold_w:
|
|
later = next(
|
|
(p for t, p in self._power_readings if t > end_time), None
|
|
)
|
|
if later is not None and later < self._config.stop_threshold_w:
|
|
last_p = later
|
|
final_readings.append((end_time, last_p))
|
|
|
|
start_ts = self._current_cycle_start.timestamp()
|
|
# Store timestamps in canonical UTC (#369). Reading timestamps arrive from
|
|
# dt_util.now() (HA-local-aware) while trim/split paths emit UTC, which left
|
|
# past_cycles with a mix of offsets. Normalizing here (instant-preserving)
|
|
# keeps stored cycles consistent and safe for cross-device/store transfer.
|
|
cycle_data: dict[str, Any] = {
|
|
"start_time": dt_util.as_utc(self._current_cycle_start).isoformat(),
|
|
"end_time": dt_util.as_utc(end_time).isoformat(),
|
|
"duration": duration,
|
|
"max_power": self._cycle_max_power,
|
|
"status": status,
|
|
"termination_reason": termination_reason,
|
|
"power_data": [[round(t.timestamp() - start_ts, 1), p] for t, p in final_readings],
|
|
}
|
|
spans = [
|
|
[round(at, 1), round(n, 1)]
|
|
for at, n in _merge_spans([*self._stall_spans, *self._user_pause_spans])
|
|
if at < duration
|
|
]
|
|
if spans:
|
|
# Item 514: the finished stalls and user pauses, so the cycle-end label
|
|
# match reads the trace in programme time (`match_rules.final_match_input`).
|
|
cycle_data["halt_spans"] = spans
|
|
|
|
self._logger.info("Cycle Finished: %s, %.1f min", status, duration / 60)
|
|
self._on_cycle_end(cycle_data)
|
|
|
|
target = STATE_FINISHED
|
|
if status == "interrupted":
|
|
target = STATE_INTERRUPTED
|
|
elif status == "force_stopped":
|
|
target = STATE_FORCE_STOPPED
|
|
elif (
|
|
status == "completed"
|
|
and termination_reason in ANTI_WRINKLE_ELIGIBLE_REASONS
|
|
and self._config.anti_wrinkle_enabled
|
|
and self._config.device_type in (
|
|
DEVICE_TYPE_WASHING_MACHINE,
|
|
DEVICE_TYPE_DRYER,
|
|
DEVICE_TYPE_WASHER_DRYER,
|
|
)
|
|
):
|
|
target = STATE_ANTI_WRINKLE
|
|
|
|
self.reset(target_state=target, timestamp=timestamp)
|
|
|
|
# Stub methods for compatibility or simpler logic
|
|
def force_end(self, timestamp: datetime) -> None:
|
|
"""Force the cycle to end immediately."""
|
|
if self._state != STATE_OFF:
|
|
self._finish_cycle(
|
|
dt_util.as_utc(timestamp),
|
|
status="force_stopped",
|
|
termination_reason=TerminationReason.FORCE_STOPPED,
|
|
keep_tail=False, # Force stop usually implies snap back to reality
|
|
)
|
|
self._ignore_power_until_idle = False
|
|
|
|
def user_stop(self) -> None:
|
|
"""Handle user-initiated stop."""
|
|
if self._state != STATE_OFF:
|
|
now = utc_now()
|
|
# "Done now" keeps the tail only while the machine is still running.
|
|
# Pressed after the wash had already gone quiet - which is exactly when
|
|
# a user stops it, because WashData missed the end - the whole wait was
|
|
# banked as cycle time (30 min late -> a 60 min wash stored as 90) and
|
|
# fed avg_duration like item 297's Smart tail. Cap it the same way
|
|
# (audit DETECT-06).
|
|
below_stop = bool(self._power_readings) and (
|
|
self._power_readings[-1][1] < self._config.stop_threshold_w
|
|
)
|
|
self._finish_cycle(
|
|
now,
|
|
status="completed",
|
|
termination_reason=TerminationReason.USER,
|
|
keep_tail=True, # User implies "Done Now"
|
|
tail_cap=(
|
|
self._keep_tail_cap(self._current_cycle_start or now)
|
|
if below_stop
|
|
else None
|
|
),
|
|
)
|
|
# Prevent immediate restart if power is still high
|
|
self._ignore_power_until_idle = True
|
|
# Anchor the lockout clock to this stop instant. The next reading's
|
|
# dt is measured from the last processed sample, which predates the
|
|
# stop, so without this the high-power accumulator would count the
|
|
# pre-stop gap and release the lockout early (#267).
|
|
self._lockout_high_seconds = 0.0
|
|
self._last_process_time = now
|
|
|
|
|
|
def get_power_trace(self) -> list[tuple[datetime, float]]:
|
|
"""Return the current power trace."""
|
|
return list(self._power_readings)
|
|
|
|
def get_state_snapshot(self) -> dict[str, Any]:
|
|
"""Get a snapshot of the current state for persistence."""
|
|
return {
|
|
"state": self._state,
|
|
"sub_state": self._sub_state,
|
|
"current_cycle_start": (
|
|
self._current_cycle_start.isoformat()
|
|
if self._current_cycle_start
|
|
else None
|
|
),
|
|
"power_readings": [(t.isoformat(), p) for t, p in self._power_readings],
|
|
"accumulated_energy_wh": self._energy_since_idle_wh,
|
|
"time_above": self._time_above_threshold,
|
|
"time_below": self._time_below_threshold,
|
|
"time_below_gapfree": self._time_below_threshold_gapfree,
|
|
"cycle_max_power": self._cycle_max_power,
|
|
"last_active_time": (
|
|
self._last_active_time.isoformat() if self._last_active_time else None
|
|
),
|
|
"expected_duration": self._expected_duration,
|
|
"matched_profile": self._matched_profile,
|
|
"state_enter_time": (
|
|
self._state_enter_time.isoformat() if self._state_enter_time else None
|
|
),
|
|
"end_spike_seen": self._end_spike_seen,
|
|
"end_spike_duration": self._end_spike_duration,
|
|
"match_ambiguous": self._match_ambiguous,
|
|
"match_prefix_ambiguous_full_shape": self._match_prefix_ambiguous_full_shape,
|
|
"matched_tail_power": self._matched_tail_power,
|
|
"matched_terminal_high": self._matched_terminal_high,
|
|
"matched_terminal_quiet_s": self._matched_terminal_quiet_s,
|
|
"matched_trusted_min_s": self._matched_trusted_min_s,
|
|
# Element 14. A dishwasher restored into its terminal-tail match freeze
|
|
# never re-matches, so without it the item-465 release floor was gone
|
|
# for the rest of the cycle (the hazard gate only lost a shortening).
|
|
"matched_pause_catalogue": (
|
|
[
|
|
self._matched_pause_catalogue[0],
|
|
[list(p) for p in self._matched_pause_catalogue[1]],
|
|
]
|
|
if self._matched_pause_catalogue is not None
|
|
else None
|
|
),
|
|
"longest_candidate_duration": self._longest_candidate_duration,
|
|
"ml_defer_start_duration": self._ml_defer_start_duration,
|
|
# Without it a restart dropped the confidence to 0.0: Smart Termination
|
|
# was then blocked as low_confidence, and a dishwasher restored into
|
|
# its terminal-tail match freeze could never re-match (DETECT-09).
|
|
"last_match_confidence": self._last_match_confidence,
|
|
# Register item 393a: the #296 tail's baseline and the idle clock of
|
|
# the anti-wrinkle state. Without the floor a restart mid-tail fell
|
|
# back to the configured burst length, which splits the tail again.
|
|
"anticrease_tail_floor_w": self._anticrease_tail_floor_w,
|
|
"anti_wrinkle_idle_time": self._anti_wrinkle_idle_time,
|
|
# Item 511: the stalls already over, which the end gates' clock
|
|
# excludes (the one in progress is re-derived from the trace).
|
|
"stall_spans": [[at, length] for at, length in self._stall_spans],
|
|
# Item 514: the user pauses already resumed from, and the current one.
|
|
"user_pause_spans": [[at, length] for at, length in self._user_pause_spans],
|
|
"user_pause_since": (
|
|
self._user_pause_since.isoformat() if self._user_pause_since else None
|
|
),
|
|
}
|
|
|
|
@property
|
|
def in_anticrease_tail(self) -> bool:
|
|
"""STATE_ANTI_WRINKLE entered by the #296 anti-crease finalise (item 393a)."""
|
|
return (
|
|
self._state == STATE_ANTI_WRINKLE
|
|
and self._anticrease_tail_floor_w is not None
|
|
)
|
|
|
|
def get_elapsed_seconds(self) -> float:
|
|
"""Return seconds elapsed in current cycle."""
|
|
if self._current_cycle_start:
|
|
return (utc_now() - self._current_cycle_start).total_seconds()
|
|
return 0.0
|
|
|
|
def is_waiting_low_power(self) -> bool:
|
|
"""Return True if we are pending end/pause due to low power."""
|
|
return (
|
|
self._state in (STATE_RUNNING, STATE_PAUSED, STATE_ENDING)
|
|
and self._time_below_threshold > 0
|
|
)
|
|
|
|
def restore_state_snapshot(self, snapshot: dict[str, Any]) -> bool:
|
|
"""Restore state from snapshot; False when it could not be restored.
|
|
|
|
A snapshot that raises leaves the detector OFF (never half-restored) and is
|
|
logged with its traceback. The caller keeps the snapshot for diagnostics
|
|
instead of deleting it (register item 266 follow-up): until then a restore
|
|
that raised lost the running cycle with one ERROR line and no record.
|
|
"""
|
|
try:
|
|
self._state = snapshot.get("state", STATE_OFF)
|
|
self._sub_state = snapshot.get("sub_state")
|
|
# Not persisted (item 501): a restored probe is shown.
|
|
self._standby_reprobe = False
|
|
self._probe_hidden = False
|
|
self._energy_since_idle_wh = snapshot.get("accumulated_energy_wh", 0.0)
|
|
self._time_above_threshold = snapshot.get("time_above", 0.0)
|
|
self._time_below_threshold = snapshot.get("time_below", 0.0)
|
|
# Old snapshots lack the gap-free tally, and the plain value they do
|
|
# carry may already include outage-sized intervals — the exact
|
|
# contamination this field exists to exclude — so it must NOT be used
|
|
# as the fallback. 0.0 is also the honest value on a restore in
|
|
# general: the restart itself is unobserved time (the manager records
|
|
# it as a restart gap), so no quiet observed before it still counts.
|
|
self._time_below_threshold_gapfree = float(
|
|
snapshot.get("time_below_gapfree", 0.0) or 0.0
|
|
)
|
|
# Not persisted: the restart's own gap is in neither tally either.
|
|
self._time_below_unobserved = 0.0
|
|
self._cycle_max_power = snapshot.get("cycle_max_power", 0.0)
|
|
# Sanitize via the same helper as update_match so the class
|
|
# invariant on _expected_duration holds across restarts and the
|
|
# gates in STATE_ENDING / _should_defer_finish can trust the value.
|
|
# If sanitization rejects the snapshot's expected_duration, also
|
|
# clear the matched_profile so we don't restore a half-valid state
|
|
# where Smart Termination can fire on _expected_duration == 0.0.
|
|
restored_match = snapshot.get("matched_profile")
|
|
sanitized_expected = self._sanitize_expected_duration(
|
|
snapshot.get("expected_duration", 0.0),
|
|
source="restore_state_snapshot",
|
|
)
|
|
if (
|
|
restored_match is not None
|
|
and sanitized_expected == self._SANITIZE_INVALID_SENTINEL
|
|
):
|
|
self._logger.debug(
|
|
"restore_state_snapshot: dropping matched_profile %r "
|
|
"because expected_duration sanitized to invalid sentinel",
|
|
restored_match,
|
|
)
|
|
self._matched_profile = None
|
|
else:
|
|
self._matched_profile = restored_match
|
|
self._expected_duration = sanitized_expected
|
|
self._end_spike_seen = snapshot.get("end_spike_seen", False)
|
|
self._end_spike_duration = float(snapshot.get("end_spike_duration", 0.0))
|
|
self._match_ambiguous = snapshot.get("match_ambiguous", False)
|
|
# A pre-#364 snapshot has no narrow flag: fall back to the old single
|
|
# `match_prefix_ambiguous` (the #364 prefix-fit flag, removed in 0.5.8,
|
|
# and before #364 the #288 verdict itself) so a restart cannot loosen
|
|
# the anti-crease gate. Newer snapshots no longer write that key.
|
|
self._match_prefix_ambiguous_full_shape = snapshot.get(
|
|
"match_prefix_ambiguous_full_shape",
|
|
snapshot.get("match_prefix_ambiguous", False),
|
|
)
|
|
self._matched_tail_power = self._sanitize_tail_power(
|
|
snapshot.get("matched_tail_power")
|
|
)
|
|
self._matched_terminal_high = self._sanitize_terminal_high(
|
|
snapshot.get("matched_terminal_high")
|
|
)
|
|
self._matched_terminal_quiet_s = self._sanitize_terminal_quiet(
|
|
snapshot.get("matched_terminal_quiet_s")
|
|
)
|
|
self._matched_trusted_min_s = self._sanitize_trusted_min(
|
|
snapshot.get("matched_trusted_min_s")
|
|
)
|
|
self._matched_pause_catalogue = self._sanitize_pause_catalogue(
|
|
snapshot.get("matched_pause_catalogue")
|
|
)
|
|
# Unconditionally, because the hazard is the value already on the
|
|
# object, not the one in the snapshot: this restores the ambiguity
|
|
# flags the ENDING gate reads, so leaving the bound untouched pairs
|
|
# them with whatever a previous cycle left behind. A snapshot written
|
|
# before this key existed yields 0.0, i.e. the old refusal.
|
|
self._longest_candidate_duration = self._sanitize_longest_candidate(
|
|
snapshot.get("longest_candidate_duration")
|
|
)
|
|
self._ml_defer_start_duration = snapshot.get("ml_defer_start_duration")
|
|
try:
|
|
self._last_match_confidence = float(
|
|
snapshot.get("last_match_confidence") or 0.0
|
|
)
|
|
except (TypeError, ValueError, OverflowError):
|
|
self._last_match_confidence = 0.0
|
|
# Item 393a. Meaningful only in the tail; an older snapshot (or any
|
|
# other state) yields None / 0.0, i.e. the configured burst length.
|
|
# The burst candidate is not restored: the restart broke the run.
|
|
in_tail = self._state == STATE_ANTI_WRINKLE
|
|
self._anticrease_tail_floor_w = (
|
|
self._sanitize_positive(snapshot.get("anticrease_tail_floor_w"))
|
|
if in_tail else None
|
|
)
|
|
self._anti_wrinkle_idle_time = (
|
|
self._sanitize_positive(snapshot.get("anti_wrinkle_idle_time")) or 0.0
|
|
if in_tail else 0.0
|
|
)
|
|
self._anti_wrinkle_candidate_start = None
|
|
self._anti_wrinkle_candidate_peak = 0.0
|
|
self._anti_wrinkle_candidate_start_power = 0.0
|
|
|
|
# Restore state enter time and recompute time_in_state from it
|
|
enter_time = snapshot.get("state_enter_time")
|
|
if enter_time:
|
|
try:
|
|
self._state_enter_time = dt_util.parse_datetime(enter_time)
|
|
if self._state_enter_time:
|
|
if self._state_enter_time.tzinfo is None:
|
|
self._state_enter_time = self._state_enter_time.replace(
|
|
tzinfo=dt_util.now().tzinfo
|
|
)
|
|
self._state_enter_time = dt_util.as_utc(self._state_enter_time)
|
|
elapsed = (utc_now() - self._state_enter_time).total_seconds()
|
|
self._time_in_state = max(0.0, elapsed)
|
|
except Exception: # pylint: disable=broad-exception-caught
|
|
self._logger.warning("Failed to parse state enter time")
|
|
|
|
start = snapshot.get("current_cycle_start")
|
|
self._current_cycle_start = None
|
|
if start:
|
|
try:
|
|
dt_start = dt_util.parse_datetime(start)
|
|
if dt_start and dt_start.tzinfo is None:
|
|
# Fix Naive Timestamp (Legacy Data)
|
|
dt_start = dt_start.replace(tzinfo=dt_util.now().tzinfo)
|
|
self._logger.warning("Restored Naive start_time, assuming local: %s", dt_start)
|
|
self._current_cycle_start = dt_util.as_utc(dt_start) if dt_start else None
|
|
except Exception: # pylint: disable=broad-exception-caught
|
|
self._logger.warning("Failed to parse start time: %s", start)
|
|
|
|
readings = snapshot.get("power_readings", [])
|
|
self._power_readings = []
|
|
|
|
# Detect naive readings once
|
|
has_naive_readings = False
|
|
|
|
for r in readings:
|
|
if isinstance(r, (list, tuple)):
|
|
reading = cast(list[Any] | tuple[Any, ...], r)
|
|
if len(reading) < 2:
|
|
continue
|
|
try:
|
|
t = dt_util.parse_datetime(str(reading[0]))
|
|
if t:
|
|
if t.tzinfo is None:
|
|
t = t.replace(tzinfo=dt_util.now().tzinfo)
|
|
has_naive_readings = True
|
|
value = float(reading[1])
|
|
if math.isfinite(value):
|
|
self._power_readings.append((dt_util.as_utc(t), value))
|
|
except (TypeError, ValueError, OverflowError) as exc:
|
|
self._logger.debug("Skipping malformed power reading %s: %s", r, exc)
|
|
|
|
if has_naive_readings:
|
|
self._logger.warning(
|
|
"Restored %d power readings with Naive timestamps (fixed to local)",
|
|
len(self._power_readings),
|
|
)
|
|
|
|
# Restore last active
|
|
last_active = snapshot.get("last_active_time")
|
|
if last_active:
|
|
dt_last = dt_util.parse_datetime(last_active)
|
|
if dt_last and dt_last.tzinfo is None:
|
|
dt_last = dt_last.replace(tzinfo=dt_util.now().tzinfo)
|
|
self._last_active_time = dt_util.as_utc(dt_last) if dt_last else None
|
|
else:
|
|
self._last_active_time = self._current_cycle_start
|
|
# #452: a stall survives a restart from the restored trace (element
|
|
# 15 is not persisted: the next match tick supplies it again).
|
|
self._stall_spans = self._sanitize_stall_spans(snapshot.get("stall_spans"))
|
|
self._user_pause_spans = self._sanitize_stall_spans(
|
|
snapshot.get("user_pause_spans")
|
|
)
|
|
_since = snapshot.get("user_pause_since")
|
|
_since_dt = dt_util.parse_datetime(_since) if isinstance(_since, str) else None
|
|
self._user_pause_since = dt_util.as_utc(_since_dt) if _since_dt else None
|
|
self._seed_stall_run()
|
|
|
|
except Exception: # pylint: disable=broad-exception-caught
|
|
self.restore_error = traceback.format_exc()
|
|
self._logger.warning(
|
|
"Could not restore the active-cycle snapshot (state %r, cycle start "
|
|
"%r); starting from OFF",
|
|
snapshot.get("state") if isinstance(snapshot, dict) else None,
|
|
snapshot.get("current_cycle_start") if isinstance(snapshot, dict) else None,
|
|
exc_info=True,
|
|
)
|
|
self.reset()
|
|
return False
|
|
self.restore_error = None
|
|
return True |