187 files

This commit is contained in:
Home Assistant Version Control
2026-10-09 11:57:31 +00:00
parent 28fa34d78a
commit 1e1b12d52b
187 changed files with 40899 additions and 32654 deletions
+421 -212
View File
@@ -20,6 +20,8 @@ from __future__ import annotations
import logging
from typing import Any, Optional
import math
import numpy as np
from .const import (
@@ -36,21 +38,22 @@ from .const import (
MATCH_DTW_RESAMPLE_N,
MATCH_DURATION_SCALE,
MATCH_DURATION_SCALE_OVERRUN,
MATCH_PREFIX_MIN_POINTS,
MATCH_PREFIX_SHAPE_MAX_RATIO,
MATCH_DURATION_WEIGHT,
MATCH_DURATION_WEIGHT_IN_PROGRESS,
MATCH_MIN_RATIO_GRACE_S,
MATCH_ENERGY_SCALE,
MATCH_ENERGY_REF_MIN_CYCLES,
MATCH_ENERGY_WEIGHT,
MATCH_KEEP_MIN_SCORE,
MATCH_MAE_PEAK_FLOOR,
MATCH_MAE_REF_PEAK,
MATCH_MAE_SCALE,
MAX_ALIGN_GRID_POINTS,
SMART_TERM_PREFIX_MAX_CANDIDATES,
SMART_TERM_PREFIX_MIN_COVERAGE,
SMART_TERM_PREFIX_MIN_POINTS,
SMART_TERM_PREFIX_MIN_RATIO,
STAGE4_INTEGRATED_ENERGY_DEVICE_TYPES,
)
from .signal_processing import integrate_wh
def stage4_energy_mode(device_type: str | None) -> str:
@@ -157,21 +160,35 @@ def find_best_alignment(
best_mae = float("inf")
final_offset = best_offset
# ~0.2 n^2 element work on a complete cycle (audit MATCH-CORE-11). It is
# byte-identical to the old ``np.mean(np.abs(c_seg - r_seg))``: the same
# subtract, abs and pairwise ``add.reduce`` divided by the count, written into
# one reused buffer. Allocating three fresh n-element arrays per offset, and
# np.mean's Python wrapper, cost more than the arithmetic. There is no exact
# shortcut for an L1 distance over shifts, and a 2-D batch over offsets was
# measured slower (the overlap length differs per offset). Any other dtype
# keeps the old expression (np.mean sums an integer array through a cast).
buf = (
np.empty(min(n_curr, n_ref))
if curr.dtype == np.float64 and ref.dtype == np.float64
else None
)
for off in range(int(min_off), int(max_off) + 1):
# intersection
c_start = max(0, off)
c_end = min(n_curr, n_ref + off)
r_start = max(0, -off)
r_end = min(n_ref, n_curr - off)
if (c_end - c_start) < 10:
length = c_end - c_start
if length < 10:
continue
r_start = max(0, -off)
c_seg = curr[c_start:c_end]
r_seg = ref[r_start:r_end]
mae = np.mean(np.abs(c_seg - r_seg))
if buf is None:
mae = np.mean(np.abs(curr[c_start:c_end] - ref[r_start:r_start + length]))
else:
seg = buf[:length]
np.subtract(curr[c_start:c_end], ref[r_start:r_start + length], out=seg)
np.abs(seg, out=seg)
mae = np.add.reduce(seg) / length
if mae < best_mae:
best_mae = mae
final_offset = off
@@ -198,8 +215,10 @@ def find_best_alignment(
corr = 0.0
# Scale-invariant MAE: express the error relative to the current cycle's
# peak (common to every candidate, so ranking is unaffected) and calibrate
# to the legacy behaviour at MATCH_MAE_REF_PEAK. See const.py for rationale.
# peak and calibrate to the legacy behaviour at MATCH_MAE_REF_PEAK (see
# const.py). On a complete cycle the peak is common to every candidate, so
# ranking is unaffected; in prefix mode the compared slice, and so its peak,
# depends on each candidate's span (deep-dive 02, audit MR-09).
current_peak = float(np.max(np.abs(curr))) if curr.size else 0.0
scaled_mae = mae * MATCH_MAE_REF_PEAK / max(current_peak, MATCH_MAE_PEAK_FLOOR)
mae_score = MATCH_MAE_SCALE / (MATCH_MAE_SCALE + scaled_mae)
@@ -250,7 +269,7 @@ def compute_dtw_lite(
for each row are written back as a single slice assignment. For the typical
matching case (n=m=200, band=0.1 → w=20, ~41 cells/row) this is ~1.9× faster
than the original element-by-element NumPy indexing loop. The anti-diagonal
vectorized fill from :func:`_dtw_cost_matrix_vectorized` is NOT used here
vectorized fill from :func:`_dtw_cost_banded` is NOT used here
because its per-diagonal Python setup overhead dominates for small n (it is
2× *slower* than the scalar loop for n=200 — the opposite of its large-n
envelope-rebuild behaviour where it wins by 1.6–8×).
@@ -317,6 +336,85 @@ def compute_dtw_lite(
_LOGGER.debug("compute_dtw_lite vectorized path failed; using scalar fallback", exc_info=True)
return _dtw_lite_scalar(xf, yf, n, m, w)
def dtw_lite_batch(x: np.ndarray, y: np.ndarray, band_width_ratio: float) -> np.ndarray:
"""``compute_dtw_lite`` for ``k`` equal-length pairs at once (rows of ``x``, ``y``).
The Stage-3 refines are all 200x200 with one band, so the ``top_n x 2`` DP
fills run as one row scan (audit MR-05): within a row, ``cell[j] = min(c[j],
local[j] + cell[j-1])`` with ``c = local + min(up, diag)`` unrolls to ``S[j] +
cummin(c[k] - S[k])`` over the row's prefix sums ``S``. ~8x faster for 10 DTWs;
the prefix sums reorder the additions, so results agree to ~1e-15 relative,
not bit for bit.
"""
xs = np.asarray(x, dtype=float)
ys = np.asarray(y, dtype=float)
k, n = xs.shape
m = ys.shape[1]
if n == 0 or m == 0:
return np.full(k, np.inf)
w = max(1, int(min(n, m) * band_width_ratio))
centers = (np.arange(1, n + 1, dtype=float) * (m / n)).astype(np.intp)
start_js = np.maximum(1, centers - w)
end_js = np.minimum(m, centers + w + 1)
prev = np.full((k, m + 1), np.inf)
prev[:, 0] = 0.0
zeros = np.zeros((k, 1))
for i in range(n):
a, b = int(start_js[i]), int(end_js[i])
local = np.abs(xs[:, i:i + 1] - ys[:, a - 1:b])
up_diag = np.minimum(prev[:, a:b + 1], prev[:, a - 1:b])
csum = np.cumsum(local, axis=1)
before = np.concatenate((zeros, csum[:, :-1]), axis=1)
row = csum + np.minimum.accumulate(up_diag - before, axis=1)
cur = np.full((k, m + 1), np.inf)
cur[:, a:b + 1] = row
prev = cur
return prev[:, m]
def _stage3_scores_batched(
pairs: list[tuple[np.ndarray, np.ndarray]],
current_peak: float,
*,
dtw_mode: str,
dtw_bandwidth: float,
l1_scale: float,
ddtw_scale: float,
ensemble_w: float,
) -> list[float]:
"""Stage-3 scores for ``scaled`` / ``ddtw`` / ``ensemble`` in one batched DTW.
``pairs`` are each candidate's (current, sample) series already on the
``MATCH_DTW_RESAMPLE_N`` grid; the arithmetic is `_dtw_component_score`'s.
"""
level = dtw_mode in ("scaled", "ensemble")
deriv = dtw_mode in ("ddtw", "ensemble")
xs: list[np.ndarray] = []
ys: list[np.ndarray] = []
if level:
xs.extend(a for a, _ in pairs)
ys.extend(b for _, b in pairs)
if deriv:
xs.extend(np.gradient(a) for a, _ in pairs)
ys.extend(np.gradient(b) for _, b in pairs)
dists = dtw_lite_batch(np.vstack(xs), np.vstack(ys), dtw_bandwidth)
peak = max(current_peak, MATCH_MAE_PEAK_FLOOR)
def _score(dist: float, scale: float) -> float:
scaled = (dist / MATCH_DTW_RESAMPLE_N) * MATCH_MAE_REF_PEAK / peak
return scale / (scale + scaled)
k = len(pairs)
if dtw_mode == "ensemble":
return [
ensemble_w * _score(float(dists[i]), l1_scale)
+ (1.0 - ensemble_w) * _score(float(dists[k + i]), ddtw_scale)
for i in range(k)
]
scale = ddtw_scale if deriv else l1_scale
return [_score(float(d), scale) for d in dists]
def _resample_to(arr: np.ndarray, n: int) -> np.ndarray:
"""Linearly resample a 1-D array to exactly ``n`` points over its index span.
@@ -364,14 +462,15 @@ def _stage3_dtw_score(
ddtw_scale: float,
ensemble_w: float,
curr_resampled: np.ndarray | None = None,
) -> tuple[float, float]:
"""``(dtw_score, norm_dist)`` for one candidate: the four-way ``dtw_mode``
branch of the Stage-3 refinement.
) -> float:
"""The DTW score for one candidate: the four-way ``dtw_mode`` branch of the
Stage-3 refinement.
Lifted verbatim out of ``compute_matches_worker`` so the Stage-3 loop and the
Stage-6 prefix pass (#364) share one implementation and cannot drift apart.
Behaviour-identical to the inlined version, including ``legacy`` mode's
``dtw_dist / len(curr_arr)`` normalisation and its ``norm_dist`` bookkeeping.
Lifted verbatim out of ``compute_matches_worker`` (for the #364 Stage-6 prefix
pass, removed in 0.5.8). Behaviour-identical to the inlined version, including ``legacy`` mode's
``dtw_dist / len(curr_arr)`` normalisation. (The normalised distance it also
returned until audit MR-12 went to a ``dtw_dist`` candidate field nothing
read, always 0.0 under the default ``ensemble``.)
"""
if dtw_mode == "legacy":
# Original behaviour: raw sequences, distance / len(current),
@@ -379,14 +478,13 @@ def _stage3_dtw_score(
dtw_dist = compute_dtw_lite(curr_arr, sample_arr, band_width_ratio=dtw_bandwidth)
n_points = len(curr_arr)
norm_dist = (dtw_dist / n_points) if n_points > 0 else 999.0
return 1.0 / (1.0 + norm_dist / MATCH_DTW_DIST_SCALE), norm_dist
return 1.0 / (1.0 + norm_dist / MATCH_DTW_DIST_SCALE)
if dtw_mode == "ensemble":
# Blend the level-based (L1) and shape-based (derivative) DTW
# scores; they are complementary signals.
s_l1 = _dtw_component_score(curr_arr, sample_arr, current_peak, dtw_bandwidth, False, l1_scale, curr_resampled=curr_resampled)
s_dd = _dtw_component_score(curr_arr, sample_arr, current_peak, dtw_bandwidth, True, ddtw_scale, curr_resampled=curr_resampled)
# composite; per-component distance not meaningful
return ensemble_w * s_l1 + (1.0 - ensemble_w) * s_dd, 0.0
return ensemble_w * s_l1 + (1.0 - ensemble_w) * s_dd
# "scaled" (default) or "ddtw": resample both onto one grid so the
# band and normalisation are consistent, then express the distance
# relative to the current peak (behaviour-neutral at
@@ -395,7 +493,16 @@ def _stage3_dtw_score(
scale = ddtw_scale if use_deriv else l1_scale
return _dtw_component_score(
curr_arr, sample_arr, current_peak, dtw_bandwidth, use_deriv, scale, curr_resampled=curr_resampled
), 0.0
)
def _rank_key(candidate: dict[str, Any]) -> float:
"""Sort key for candidate ranking: a non-finite score ranks last, never first."""
try:
score = float(candidate.get("score", 0.0))
except (TypeError, ValueError, OverflowError):
return float("-inf")
return score if math.isfinite(score) else float("-inf")
def compute_matches_worker(
@@ -419,6 +526,10 @@ def compute_matches_worker(
keep_min = float(config.get("keep_min_score", MATCH_KEEP_MIN_SCORE))
corr_weight = float(config.get("corr_weight", MATCH_CORR_WEIGHT))
dur_weight = float(config.get("duration_weight", MATCH_DURATION_WEIGHT))
if config.get("in_progress"):
dur_weight = float(
config.get("duration_weight_in_progress", MATCH_DURATION_WEIGHT_IN_PROGRESS)
)
en_weight = float(config.get("energy_weight", MATCH_ENERGY_WEIGHT))
dur_scale = float(config.get("duration_scale", MATCH_DURATION_SCALE))
dur_overrun_scale = float(
@@ -432,6 +543,7 @@ def compute_matches_worker(
# match path only) so the final match at cycle end - where the whole-cycle
# figures are the right comparison - is byte-identical.
in_progress = bool(config.get("in_progress"))
min_ratio_grace_s = float(config.get("min_ratio_grace_s", MATCH_MIN_RATIO_GRACE_S))
# ...and Stages 2/3 score the SHAPE against the same truncated stretch, while the
# cycle is still clearly mid-run (MATCH_PREFIX_SHAPE_MAX_RATIO). On by default
# for a live match; `prefix_shape: False` turns it off for the A/B harnesses.
@@ -447,10 +559,14 @@ def compute_matches_worker(
profile_duration = item["avg_duration"]
sample_power = item["sample_power"]
# Duration Check
# Duration Check. The lower bound waits out MATCH_MIN_RATIO_GRACE_S of a live
# match: early on it rejects every programme longer than the cycle is old.
if profile_duration > 0:
ratio = current_duration / profile_duration
if ratio < min_duration_ratio or ratio > max_duration_ratio:
lower_applies = not (
in_progress and current_duration < min_ratio_grace_s
)
if (lower_applies and ratio < min_duration_ratio) or ratio > max_duration_ratio:
continue
# Core Similarity. While the cycle is mid-run this compares it against the
@@ -468,12 +584,14 @@ def compute_matches_worker(
shape_pair = prefix_shape_arrays(
curr_arr, sample_power, current_duration, span_s
)
# The alignment offset is not kept: nothing read it, and it was in different
# units on the two paths (shared-grid index vs native index; audit MR-12).
if shape_pair is not None:
score, metrics, offset = find_best_alignment(
score, metrics, _offset = find_best_alignment(
shape_pair[0], shape_pair[1], 1.0, corr_weight=corr_weight
)
else:
score, metrics, offset = find_best_alignment(
score, metrics, _offset = find_best_alignment(
current_power, sample_power, 1.0, corr_weight=corr_weight
)
@@ -490,12 +608,11 @@ def compute_matches_worker(
"_shape_pair": shape_pair,
# True wall-clock span of `sample`, for prefix truncation (#364).
# Falls back to profile_duration so the other snapshot builders
# (devtools, matching_tuner, playground) keep working unchanged.
# (devtools, playground) keep working unchanged.
"sample_span_s": float(item.get("sample_span_s") or profile_duration or 0.0),
"offset": offset
})
candidates.sort(key=lambda x: x["score"], reverse=True)
candidates.sort(key=_rank_key, reverse=True)
# Stage 3: DTW Refinement on the top N candidates
if dtw_bandwidth > 0.0 and len(candidates) > 0:
@@ -512,7 +629,35 @@ def compute_matches_worker(
# Resample the current trace once — it's the same for every candidate.
curr_resampled = _resample_to(curr_arr, MATCH_DTW_RESAMPLE_N)
for cand in to_refine:
batched: list[float] | None = None
if dtw_mode in ("scaled", "ddtw", "ensemble") and to_refine:
# One batched DTW for every refine (audit MR-05: Stage 3 was ~89% of
# matcher CPU). Falls back to the per-candidate path on any error.
try:
grid_pairs = []
for cand in to_refine:
pair = cand.get("_shape_pair")
if pair is not None:
a = _resample_to(pair[0], MATCH_DTW_RESAMPLE_N)
b = _resample_to(pair[1], MATCH_DTW_RESAMPLE_N)
else:
a = curr_resampled
b = _resample_to(np.array(cand["sample"]), MATCH_DTW_RESAMPLE_N)
grid_pairs.append((a, b))
batched = _stage3_scores_batched(
grid_pairs, current_peak, dtw_mode=dtw_mode,
dtw_bandwidth=dtw_bandwidth, l1_scale=l1_scale,
ddtw_scale=ddtw_scale, ensemble_w=ensemble_w,
)
except Exception: # pylint: disable=broad-exception-caught
_LOGGER.debug("batched Stage-3 DTW failed; per-candidate path", exc_info=True)
batched = None
for idx, cand in enumerate(to_refine):
if batched is not None:
cand["original_score"] = float(cand["score"])
cand["score"] = float(blend * cand["score"] + (1.0 - blend) * batched[idx])
continue
pair = cand.get("_shape_pair")
if pair is not None:
warp_curr, sample_arr, cand_resampled = pair[0], pair[1], None
@@ -521,7 +666,7 @@ def compute_matches_worker(
curr_arr, np.array(cand["sample"]), curr_resampled
)
dtw_score, norm_dist = _stage3_dtw_score(
dtw_score = _stage3_dtw_score(
warp_curr,
sample_arr,
current_peak,
@@ -535,9 +680,8 @@ def compute_matches_worker(
cand["original_score"] = float(cand["score"])
cand["score"] = float(blend * cand["score"] + (1.0 - blend) * dtw_score)
cand["dtw_dist"] = float(norm_dist)
candidates.sort(key=lambda x: x["score"], reverse=True)
candidates.sort(key=_rank_key, reverse=True)
# The truncated pair is scratch for the two shape stages; it must not reach the
# MatchResult ranking (numpy arrays, and the store/WS serialise that dict).
@@ -565,6 +709,10 @@ def compute_matches_worker(
# "integrated" compares true integrated energy (mean x duration). Opt-in so
# the historical default is byte-for-byte preserved. See register item 99.
integrated = config.get("energy_mode", "mean") == "integrated"
own_energy = (
{str(s.get("name")): s.get("energy_ref") for s in snapshots}
if integrated and not in_progress else {}
)
cur_mean = float(np.mean(curr_arr))
cur_energy = cur_mean * current_duration if integrated else cur_mean
for cand in candidates:
@@ -597,6 +745,17 @@ def compute_matches_worker(
# time as cur_energy does, so a prefix mean is scaled by the elapsed
# duration and a whole-template mean by the candidate's own duration.
cand_energy = cand_mean * cand_span if integrated else cand_mean
if integrated and not in_progress:
# A COMPLETED cycle is graded against the median energy of the
# profile's own cycles, not mean(template) x duration: on a warped
# envelope that inherits the reference cycle's heating length (w7g).
# Not while running: rescaling the prefix by median/template gained
# 0.8-1.6pp top-1 at 25-75% elapsed, but the live matches it moved
# are the ones the end gates read (end_gate_eval --loo: 2 new ends
# > 5 min early, washer-dryer median lag +0.7 min).
own = own_energy_ws(own_energy.get(str(cand["name"])))
if own is not None:
cand_energy = own
en_ag = _agreement(cur_energy, cand_energy, en_scale)
cand["shape_score"] = float(cand["score"])
cand["score"] = float(
@@ -604,14 +763,17 @@ def compute_matches_worker(
+ dur_w * dur_ag
+ en_w * en_ag
)
candidates.sort(key=lambda x: x["score"], reverse=True)
candidates.sort(key=_rank_key, reverse=True)
# Stage 6 (#364): prefix scores for the few candidates materially LONGER than
# the winner. Purely additive - it writes `prefix_score` and never touches
# `score`, so ranking is provably unchanged. Must run after the Stage-4
# re-sort because the anchor is the winner's duration.
annotate_prefix_scores(candidates, curr_arr, current_duration, config)
# A non-finite score (a NaN in an imported template, a NaN avg_duration) used to
# sort to rank 1 with a NaN margin that was never "ambiguous" (audit
# MATCH-CORE-05). It carries no evidence: drop it.
candidates[:] = [c for c in candidates if math.isfinite(_rank_key(c))]
# (Stage 6, the #364 prefix scores for the Smart-Termination guard, was removed
# in 0.5.8: Stages 2/3 already score a running cycle on each candidate's
# truncated curve, so the guard fired on 0 of 713 genuine ends and 0 of 7
# quiet split positives on the shipped matcher - devtools/prefix_guard_eval.py.)
return candidates
def prefix_mean(
@@ -630,7 +792,7 @@ def prefix_mean(
Falls back to the whole template (and the candidate's own duration) when the
cycle has already outlasted it: that candidate has finished, so its total is
the honest comparison. Deliberately NOT ``_prefix_point_count``: that helper's
12-sample floor exists because Stage 6 *correlates* the prefix, while a mean
12-sample floor exists because Stages 2/3 *correlate* the prefix, while a mean
over a handful of leading samples is perfectly well defined - applying the
floor here would silently restore whole-template energy for the first few
percent of every cycle, which is exactly the window #400 is about.
@@ -643,6 +805,69 @@ def prefix_mean(
return float(np.mean(sample)), profile_duration
def member_energy_reference(
members: list[tuple[list[float], list[float], float]],
) -> dict[str, Any] | None:
"""A profile's energy taken from its OWN cycles, for Stage 4.
``members`` are the ``(offsets, watts, duration)`` triples the envelope is built
from. Each member's energy is measured the way Stage 4 measures the cycle being
matched - mean of the linearly interpolated trace x duration - so both sides of
the agreement are the same quantity. ``{"n": members used, "median_wh": median
whole-cycle Wh}``, or None without a usable member.
Why (w7g): Stage 4 took a profile's energy as ``mean(template) x avg_duration``.
On a DTW-warped envelope that inherits the heating length of the cycle the
members were warped onto: > 20% off its own members' median on 17 of 85 washer
profiles in the corpus, 2.5x on one.
"""
totals: list[float] = []
for offsets, watts, duration in members:
t = np.asarray(offsets, dtype=float)
p = np.asarray(watts, dtype=float)
if t.size != p.size or t.size < 2:
continue
ok = np.isfinite(t) & np.isfinite(p)
t, p = t[ok], p[ok]
if t.size < 2:
continue
order = np.argsort(t, kind="mergesort")
t, p = t[order], p[order]
span = float(t[-1] - t[0])
if span <= 0:
continue
try:
dur = float(duration)
except (TypeError, ValueError, OverflowError):
dur = 0.0
if not math.isfinite(dur) or dur <= 0:
dur = span
energy_ws = integrate_wh(t, p) * 3600.0 / span * dur
if math.isfinite(energy_ws) and energy_ws > 0:
totals.append(energy_ws)
if not totals:
return None
return {"n": len(totals), "median_wh": round(float(np.median(totals)) / 3600.0, 3)}
def own_energy_ws(ref: dict[str, Any] | None) -> float | None:
"""The whole-cycle energy (W*s) Stage 4 expects from a profile's own cycles.
The median of a :func:`member_energy_reference`, or None (keep the template's
``mean x duration``) without one of at least ``MATCH_ENERGY_REF_MIN_CYCLES``
cycles or with a malformed one.
"""
if not isinstance(ref, dict):
return None
try:
if int(ref.get("n") or 0) < MATCH_ENERGY_REF_MIN_CYCLES:
return None
own_ws = float(ref["median_wh"]) * 3600.0
except (TypeError, ValueError, KeyError, OverflowError):
return None
return own_ws if math.isfinite(own_ws) and own_ws > 0 else None
def _prefix_point_count(
n_points: int, current_duration: float, sample_span_s: float
) -> int:
@@ -652,12 +877,12 @@ def _prefix_point_count(
the whole template (then it is not a prefix), or when too few points remain to
judge. Fraction-of-array is the right operator because every snapshot flavour
is uniform in time over its own span (envelope: np.linspace; sample cycle:
resample_uniform at a fixed dt; group aggregate: np.interp onto 200 points).
resample_uniform at a fixed dt). (The group aggregate snapshot is gone, #400.)
"""
if n_points < SMART_TERM_PREFIX_MIN_POINTS or sample_span_s <= 0 or current_duration <= 0:
if n_points < MATCH_PREFIX_MIN_POINTS or sample_span_s <= 0 or current_duration <= 0:
return 0
k = int(round(n_points * (current_duration / sample_span_s)))
if k < SMART_TERM_PREFIX_MIN_POINTS or k >= n_points:
if k < MATCH_PREFIX_MIN_POINTS or k >= n_points:
return 0
return k
@@ -671,140 +896,38 @@ def prefix_shape_arrays(
"""``(current, template)`` on one grid, with the template TRUNCATED to the
elapsed time - or None when it cannot be truncated meaningfully.
One definition of "the same stretch of both curves", shared by the two callers
that need it: the live Stage-2/3 shape scoring (#400) and the Stage-6 prefix
guard (#364). Both series go onto a shared grid so an index offset equals a time
offset regardless of the template's native cadence; the grid also honours the
#388 OOM cap. The 12-sample floor is real here (unlike in ``prefix_mean``):
these arrays get correlated and warped, not averaged.
One definition of "the same stretch of both curves" for the live Stage-2/3
shape scoring (#400). (It also fed the #364 Stage-6 prefix guard, removed in
0.5.8; ``devtools/prefix_guard_eval.py`` keeps a copy of that rule.) Both
series go onto a shared grid so an index offset equals a time offset
regardless of the template's native cadence; the grid also honours the #388
OOM cap. The 12-sample floor is real here (unlike in ``prefix_mean``): these
arrays get correlated and warped, not averaged.
The grid is shared **between the two series**, not across candidates: ``k``
is the candidate's own truncated point count, so a longer template can be
scored on a finer grid than a shorter one. That asymmetry is deliberate and
measured. Capping the grid at ``k`` is what stops ``arr[:k]`` being upsampled
past the points it actually has, which would invent template detail the
recording never contained. Dropping the ``k`` term to make the grid purely
candidate-independent (``min(curr_arr.size, MAX_ALIGN_GRID_POINTS)``) was
tried and measured on ``devtools/prefix_guard_eval.py``: at the shipped
constants it takes the #364 split guard from **59/114 caught (52%) to 52/114
(46%)** while removing only 2 of 13 false blocks. A miss there is a SPLIT
CYCLE and a false block is merely a later finish, so that trade is
net-negative. ``devtools/dtw_ab_eval.py`` is byte-identical either way
(it scores only complete cycles, which never take the prefix path), so it
cannot be used to judge this function - use ``prefix_guard_eval.py``.
scored on a finer grid than a shorter one. Capping the grid at ``k`` is what
stops ``arr[:k]`` being upsampled past the points it actually has, which
would invent template detail the recording never contained. (The one A/B of
dropping the ``k`` term was measured on the removed guard, not on matching.)
``devtools/dtw_ab_eval.py`` scores only complete cycles, which never take
the prefix path, so it cannot judge this function - use ``devtools/eval.py``.
"""
arr = np.asarray(sample, dtype=float)
k = _prefix_point_count(arr.size, current_duration, sample_span_s)
if k == 0:
return None
grid = int(min(curr_arr.size, k, MAX_ALIGN_GRID_POINTS))
if grid < SMART_TERM_PREFIX_MIN_POINTS:
if grid < MATCH_PREFIX_MIN_POINTS:
return None
return _resample_to(curr_arr, grid), _resample_to(arr[:k], grid)
def prefix_shape_score(
curr_arr: np.ndarray,
sample: list[float] | np.ndarray,
current_duration: float,
sample_span_s: float,
current_peak: float,
config: dict[str, Any],
) -> float | None:
"""Score the live trace against ``sample`` TRUNCATED to ``current_duration``.
The #288 landscape guard asks whether a longer candidate has a decent shape
score against its **whole** curve - which a part-way-through trace cannot
have. This asks the question that actually matters: does the trace look like
the *beginning* of that longer programme? (#364)
Same scale as ``shape_score`` by construction: identical Stage-2 formula
(``find_best_alignment``) and identical Stage-3 DTW blend, only the reference
array differs. Returns None when the template cannot be truncated meaningfully.
NB prefix scoring normalizes on the shared resample ``grid`` (both series are
resampled to it), so it does not support the non-default ``dtw_mode="legacy"``
absolute-watt/length normalization - under which cross-candidate prefix scores of
differing native length would not be comparable. This is inert in production: the
default is ``"ensemble"`` and the live ProfileStore path never sets ``dtw_mode``;
``"legacy"`` exists only for the devtools re-sweep harness.
"""
pair = prefix_shape_arrays(curr_arr, sample, current_duration, sample_span_s)
if pair is None:
return None
a, b = pair
corr_weight = float(config.get("corr_weight", MATCH_CORR_WEIGHT))
score, _metrics, _offset = find_best_alignment(a, b, 1.0, corr_weight=corr_weight)
# The SAME default as `compute_matches_worker` (register item 309). Both read
# the same unmutated `config` in one match, so a caller that omits the key
# would otherwise get DEFAULT_DTW_BANDWIDTH for Stage 3 and the old 0.1 here,
# and the two stages would disagree about which candidates look like a prefix.
dtw_bandwidth = float(config.get("dtw_bandwidth", DEFAULT_DTW_BANDWIDTH))
if dtw_bandwidth > 0.0:
dtw_score, _ = _stage3_dtw_score(
a,
b,
current_peak,
dtw_mode=str(config.get("dtw_mode", DEFAULT_DTW_MODE)),
dtw_bandwidth=dtw_bandwidth,
l1_scale=float(config.get("dtw_l1_scale", MATCH_DTW_DIST_SCALE)),
ddtw_scale=float(config.get("dtw_ddtw_scale", MATCH_DDTW_DIST_SCALE)),
ensemble_w=float(config.get("dtw_ensemble_w", MATCH_DTW_ENSEMBLE_W)),
)
blend = float(config.get("dtw_blend", MATCH_DTW_BLEND))
return float(blend * score + (1.0 - blend) * dtw_score)
return float(score)
def annotate_prefix_scores(
candidates: list[dict[str, Any]],
curr_arr: np.ndarray,
current_duration: float,
config: dict[str, Any],
) -> None:
"""Stage 6 (#364): write ``prefix_score`` on the few non-winning candidates
that are materially longer than the winner.
Mutates in place and never touches ``score``/``shape_score``, so candidate
ranking is unaffected - this only feeds the Smart-Termination prefix guard.
Every test before the first array touch is a scalar compare, so the common
case (no candidate is materially longer) costs nothing.
"""
if current_duration <= 0 or len(candidates) < 2 or curr_arr.size == 0:
return
best_dur = float(candidates[0].get("profile_duration") or 0.0)
if best_dur <= 0:
return
min_dur = best_dur * SMART_TERM_PREFIX_MIN_RATIO
current_peak = float(np.max(curr_arr))
scored = 0
for cand in candidates[1:]:
prof_dur = float(cand.get("profile_duration") or 0.0)
if prof_dur <= min_dur:
continue # not a longer look-alike
if prof_dur <= current_duration:
continue # we already outlasted it, so we are not inside its prefix
span = float(cand.get("sample_span_s") or prof_dur)
if span < prof_dur * SMART_TERM_PREFIX_MIN_COVERAGE:
continue # gap-truncated template: may not start at the programme's start
score = prefix_shape_score(
curr_arr, cand.get("sample") or [], current_duration, span, current_peak, config
)
if score is None:
continue
cand["prefix_score"] = float(score)
scored += 1
if scored >= SMART_TERM_PREFIX_MAX_CANDIDATES:
break
def _dtw_cost_matrix_scalar(
x: np.ndarray, y: np.ndarray, n: int, m: int, w: int
) -> np.ndarray:
"""Reference (scalar) Sakoe-Chiba DTW cost-matrix fill. Kept verbatim as the
fallback for :func:`_dtw_cost_matrix_vectorized` so behavior can never regress."""
fallback for :func:`_dtw_cost_banded` so behavior can never regress."""
cost_matrix = np.full((n + 1, m + 1), float("inf"))
cost_matrix[0, 0] = 0
for i in range(1, n + 1):
@@ -819,48 +942,106 @@ def _dtw_cost_matrix_scalar(
return cost_matrix
def _dtw_cost_matrix_vectorized(
class _BandedCostMatrix:
"""A Sakoe-Chiba DTW cost matrix that stores only its in-band cells.
Cells are kept per anti-diagonal ``d = i + j`` (the order the fill computes
them), each diagonal padded with one ``inf`` cell on either side. ``get``
returns exactly what the full ``(n+1) x (m+1)`` matrix held at ``(i, j)``:
the computed value in band, ``inf`` everywhere else (audit LIVE-20).
"""
__slots__ = ("_vals", "_first", "_count", "_start")
def __init__(self, vals: np.ndarray, first: list[int], count: list[int], start: list[int]) -> None:
self._vals = vals
self._first = first
self._count = count
self._start = start
def get(self, i: int, j: int) -> float:
d = i + j
k = i - self._first[d]
if 0 <= k < self._count[d]:
# .item(): a Python float, without boxing the whole array (a .tolist()
# copy of a 2000-point band costs ~4x the band itself).
return self._vals.item(self._start[d] + 1 + k)
return math.inf
def _dtw_cost_banded(
x: np.ndarray, y: np.ndarray, n: int, m: int, w: int
) -> np.ndarray:
"""Bit-identical vectorized fill of the scalar cost matrix.
) -> _BandedCostMatrix:
"""Bit-identical vectorized fill of the scalar cost matrix, in-band cells only.
The DTW recurrence is sequential, but all cells on one anti-diagonal
(``i + j`` constant) depend only on earlier anti-diagonals, so each diagonal
is one vectorized NumPy update instead of thousands of Python ``min``/``abs``
calls. The Sakoe-Chiba band, the per-row bounds (``int`` truncation), the
``local + min(up, left, diag)`` recurrence and out-of-band ``inf`` cells all
match the scalar loop exactly, so the resulting matrix - and the backtracked
path - is identical. (#311 follow-up: this fill dominates envelope rebuilds.)
(``i + j`` constant) depend only on the two before it, so each diagonal is
one vectorized NumPy update. The Sakoe-Chiba band, the per-row bounds
(``int`` truncation), the ``local + min(min(up, left), diag)`` recurrence and
the out-of-band ``inf`` cells match the scalar loop exactly, so every cell -
and the backtracked path - is identical. (#311 follow-up: this fill
dominates envelope rebuilds.)
Until audit LIVE-20 it filled a full ``(n+1) x (m+1)`` matrix and masked all
of every anti-diagonal down to the band. Both band edges are non-decreasing
in ``i``, so the in-band cells of a diagonal are one contiguous run of rows,
found by a binary search; and a run moves by at most one row from one
diagonal to the next, so one ``inf`` pad cell per side makes every
predecessor lookup a slice. A 2000 x 2000 pair at the default 20% band
stores ~2.5x fewer cells and fills ~4-8x faster.
"""
xf = np.asarray(x, dtype=float)
yf = np.asarray(y, dtype=float)
cost_matrix = np.full((n + 1, m + 1), np.inf)
cost_matrix[0, 0] = 0.0
# Per-row band bounds, identical to the scalar start_j/end_j (int truncates
# toward zero, matching Python int()).
i_idx = np.arange(1, n + 1)
center = i_idx * (m / n)
lo = np.maximum(1, (center - w).astype(np.int64))
hi = np.minimum(m, (center + w).astype(np.int64) + 1)
# Rows in band on diagonal d: i + lo[i] <= d <= i + hi[i] (both sides strictly
# increasing in i), inside the matrix (1 <= i <= n, 1 <= d - i <= m).
d_all = np.arange(n + m + 1)
first = np.searchsorted(i_idx + hi, d_all, side="left") + 1
last = np.searchsorted(i_idx + lo, d_all, side="right")
first = np.maximum(first, np.maximum(1, d_all - m))
last = np.minimum(last, np.minimum(n, d_all - 1))
count = np.maximum(0, last - first + 1)
first[0], count[0] = 0, 1 # the origin (0, 0)
start = np.zeros(n + m + 2, dtype=np.int64)
np.cumsum(count + 2, out=start[1:])
vals = np.full(int(start[-1]), np.inf)
vals[1] = 0.0
first_l: list[int] = first.tolist()
count_l: list[int] = count.tolist()
start_l: list[int] = start.tolist()
def run(e: int, p: int, q: int) -> np.ndarray:
"""Cells of rows ``p..q`` on diagonal ``e`` (``inf`` outside its band)."""
fe, ce = first_l[e], count_l[e]
base = start_l[e] + 1 - fe
if p >= fe - 1 and q <= fe + ce:
return vals[base + p: base + q + 1]
out = np.full(q - p + 1, np.inf)
a, b = max(p, fe), min(q, fe + ce - 1)
if a <= b:
out[a - p: b - p + 1] = vals[base + a: base + b + 1]
return out
for d in range(2, n + m + 1):
i_lo = max(1, d - m)
i_hi = min(n, d - 1)
if i_lo > i_hi:
c = count_l[d]
if c <= 0:
continue
ii = np.arange(i_lo, i_hi + 1)
jj = d - ii
inb = (jj >= lo[ii - 1]) & (jj <= hi[ii - 1])
if not inb.any():
continue
ib = ii[inb]
jb = jj[inb]
local = np.abs(xf[ib - 1] - yf[jb - 1])
p = first_l[d]
q = p + c - 1
# cells (i, d - i) for i = p..q: x[i - 1] against y[d - i - 1]
local = np.abs(xf[p - 1: q] - yf[d - q - 1: d - p][::-1])
best = np.minimum(
np.minimum(cost_matrix[ib - 1, jb], cost_matrix[ib, jb - 1]),
cost_matrix[ib - 1, jb - 1],
np.minimum(run(d - 1, p - 1, q - 1), run(d - 1, p, q)), # up, left
run(d - 2, p - 1, q - 1), # diag
)
cost_matrix[ib, jb] = local + best
return cost_matrix
s = start_l[d] + 1
np.add(local, best, out=vals[s: s + c])
return _BandedCostMatrix(vals, first_l, count_l, start_l)
def compute_dtw_path(
@@ -874,11 +1055,13 @@ def compute_dtw_path(
if n == 0 or m == 0:
return []
# Pre-flight memory guard: the cost matrix is (n+1)x(m+1) float64. An
# uncapped call from a 1 Hz long cycle can request >1 GB here. If the
# allocation would exceed ~80 MB, skip DTW and return an empty path so
# Pre-flight memory guard: the scalar fallback's cost matrix is (n+1)x(m+1)
# float64. An uncapped call from a 1 Hz long cycle can request >1 GB there.
# If the allocation would exceed ~80 MB, skip DTW and return an empty path so
# the caller falls back to linear interpolation (graceful degrade rather
# than OOM-killing Home Assistant — issue #388).
# than OOM-killing Home Assistant, issue #388). The banded fill needs far
# less, but the cap stays where it was so which pairs get a path is unchanged
# (``_closed_end_path_exists`` mirrors it).
_DTW_CELL_BUDGET = 10_000_000 # 10 M cells x 8 B ≈ 80 MB
if (n + 1) * (m + 1) > _DTW_CELL_BUDGET:
_LOGGER.warning(
@@ -890,12 +1073,13 @@ def compute_dtw_path(
w = max(1, int(min(n, m) * band_width_ratio))
try:
cost_matrix = _dtw_cost_matrix_vectorized(x, y, n, m, w)
cost = _dtw_cost_banded(x, y, n, m, w).get
except Exception: # pylint: disable=broad-exception-caught
cost_matrix = _dtw_cost_matrix_scalar(x, y, n, m, w)
full = _dtw_cost_matrix_scalar(x, y, n, m, w)
cost = lambda i, j: full[i, j] # noqa: E731
# Backtracking
if np.isinf(cost_matrix[n, m]):
if math.isinf(cost(n, m)):
# Endpoint is unreachable (e.g. Sakoe-Chiba band excluded it); no valid path.
return []
@@ -912,9 +1096,9 @@ def compute_dtw_path(
i -= 1
else:
candidates_cost = [
(cost_matrix[i - 1, j], 0), # deletion (i-1)
(cost_matrix[i, j - 1], 1), # insertion (j-1)
(cost_matrix[i - 1, j - 1], 2) # match (both)
(cost(i - 1, j), 0), # deletion (i-1)
(cost(i, j - 1), 1), # insertion (j-1)
(cost(i - 1, j - 1), 2) # match (both)
]
candidates_cost.sort(key=lambda item: item[0])
best_move = candidates_cost[0][1]
@@ -967,7 +1151,7 @@ def compute_envelope_worker(
try:
offsets_list, values_list, *rest = curve
curve_duration = rest[0] if rest else None
except (ValueError, TypeError):
except (ValueError, TypeError, OverflowError):
continue
if not offsets_list or not values_list:
@@ -986,7 +1170,7 @@ def compute_envelope_worker(
try:
offsets = np.asarray(offsets_list, dtype=float)
values = np.asarray(values_list, dtype=float)
except (TypeError, ValueError):
except (TypeError, ValueError, OverflowError):
continue
# Drop paired entries where either coordinate is non-finite.
@@ -1202,7 +1386,8 @@ def align_trace_to_envelope(
**Known limit, measured and accepted (register item 347).** The observed span
here is ``t_obs[-1]``, while ``_rebuild_envelope_sync`` builds each member
over ``manual_duration or max(last_offset, stored_duration)``. Where the
over ``manual_duration`` (when plausible for its trace, audit MATCH-EVAL-17)
or ``max(last_offset, stored_duration)``. Where the
stored duration runs past the last sample the build covered a slightly longer
span than this re-derivation does, so the warp is not bit-for-bit the build's
own. Measured over the 703 stored cycles in the maintainer's corpus it
@@ -1285,6 +1470,28 @@ def align_trace_to_envelope(
return _proportional(), False
_DTW_PATH_CELL_BUDGET = 10_000_000 # compute_dtw_path's matrix cap
def _closed_end_path_exists(n: int, m: int, band_width_ratio: float) -> bool:
"""Whether ``compute_dtw_path`` would return a path for an ``n x m`` pair.
Exactly its three failure cases without filling the matrix: an empty side,
the cell budget, and an end cell the Sakoe-Chiba band cannot reach. Each row's
band is ``[lo, hi]`` (the fill's own truncation); cells are reachable left to
right from the diagonal, so the end is reachable iff no row's band starts more
than one column past where the previous row's ends (row 0 is the origin).
"""
if n == 0 or m == 0 or (n + 1) * (m + 1) > _DTW_PATH_CELL_BUDGET:
return False
w = max(1, int(min(n, m) * band_width_ratio))
center = np.arange(1, n + 1) * (m / n)
lo = np.maximum(1, (center - w).astype(np.int64))
hi = np.minimum(m, (center + w).astype(np.int64) + 1)
prev_hi = np.concatenate(([0], hi[:-1]))
return bool(np.all(lo <= prev_hi + 1) and np.all(lo <= hi) and hi[-1] == m)
def verify_profile_alignment_worker(
current_power: list[float],
envelope_avg_curve: list[float],
@@ -1321,17 +1528,19 @@ def verify_profile_alignment_worker(
if offset < 0:
curr_seg = curr[-offset:]
path = compute_dtw_path(curr_seg, ref_seg, band_width_ratio=dtw_bandwidth)
if not path:
# The DTW that used to run here was closed-end: its backtrack starts at the last
# cell, so the mapped index was ALWAYS the window's last index whenever a path
# existed - 34x the CPU and a 33-42 MiB matrix per tick for a constant (audit
# LIVE-03). Same answer, same fallback, no matrix: a path exists unless the
# window is empty, over the old cell budget, or the band cannot reach the end
# cell. The resulting position runs ALIGNMENT_CONTEXT_BUFFER // 2 steps ahead
# of the trace; measured harmless-to-helpful, so it is kept, not "fixed".
if _closed_end_path_exists(len(curr_seg), len(ref_seg), dtw_bandwidth):
mapped_idx = end_ref - 1
else:
# Fallback to linear mapping based on offset
mapped_idx = min(len(ref)-1, offset + len(curr) - 1)
mapped_idx = max(0, mapped_idx)
else:
# Map the final point of the current trace to the reference index
last_pair = path[-1]
ref_seg_idx = last_pair[1]
mapped_idx = start_ref + ref_seg_idx
# Ensure sequences are non-empty before indexing
if not envelope_time_grid or len(ref) == 0: