393 files

This commit is contained in:
Home Assistant Version Control
2026-09-27 17:16:54 +00:00
parent b05b7a897e
commit 39d97ac2db
404 changed files with 54006 additions and 11034 deletions
@@ -36,37 +36,60 @@ from typing import Any
import numpy as np
from .. import analysis
from ..const import (
DEFAULT_DTW_BANDWIDTH,
DEFAULT_DTW_MODE,
DEFAULT_PROFILE_MATCH_MAX_DURATION_RATIO,
DEFAULT_PROFILE_MATCH_MIN_DURATION_RATIO,
)
from ..signal_processing import resample_adaptive, resample_uniform
_RESAMPLE_L = 150
#: Matches the production matcher (``profile_store.async_match_profile``).
_MIN_DT = 5.0
_GAP_S = 21600.0
def _powers(cycle: dict[str, Any]) -> list[float]:
def _series(cycle: dict[str, Any]) -> tuple[np.ndarray, np.ndarray] | None:
"""``(offsets_s, watts)`` from a stored cycle, or None if unusable.
The offsets matter: ``analysis.find_best_alignment`` compares the two curves
**index by index** (its ``dt`` argument is explicitly unused), so both sides
must be on the same seconds-per-sample grid or the MAE compares different
moments of the cycle. Dropping the offsets - as this module did until register
item 303 - makes that impossible to honour.
"""
pd = cycle.get("power_data") or []
out: list[float] = []
ts: list[float] = []
pw: list[float] = []
for p in pd:
# Both values are converted BEFORE either is appended: a row whose offset
# parses and whose power does not would otherwise leave `ts` one element
# longer than `pw`, and `resample_uniform` raises inside `np.interp` on
# unequal arrays - a tuning run lost to one malformed sample.
# OverflowError too: `json` keeps an oversized integer literal as an
# unbounded int, and `float()` on one raises rather than returning inf.
try:
out.append(float(p[1]))
except (TypeError, ValueError, IndexError):
pass
return out
offset, watts = float(p[0]), float(p[1])
except (TypeError, ValueError, IndexError, OverflowError):
continue
ts.append(offset)
pw.append(watts)
if len(pw) < 4:
return None
return np.asarray(ts, dtype=float), np.asarray(pw, dtype=float)
def _resample(vals: list[float], n: int) -> np.ndarray:
a = np.asarray(vals, dtype=float)
if a.size == 0:
return np.zeros(n)
if a.size == n:
return a
return np.interp(np.linspace(0, 1, n), np.linspace(0, 1, a.size), a)
def _longest(segments: list[Any]) -> Any | None:
return max(segments, key=lambda s: len(s.power)) if segments else None
def _prep(cycles: list[dict[str, Any]]) -> dict[str, list[dict[str, Any]]]:
"""Group labelled cycles by profile, caching powers/duration/resampled curve."""
"""Group labelled cycles by profile, keeping the raw time series."""
by_profile: dict[str, list[dict[str, Any]]] = {}
for c in cycles:
name = c.get("profile_name")
pw = _powers(c)
if not name or len(pw) < 4:
series = _series(c)
if not name or series is None:
continue
try:
dur = float(c.get("duration"))
@@ -78,50 +101,216 @@ def _prep(cycles: list[dict[str, Any]]) -> dict[str, list[dict[str, Any]]]:
# sample every 30-60 s. Real cycles always carry a 'duration', so this only
# drops degenerate entries.
continue
ts, pw = series
review = c.get("ml_review")
golden = bool(review.get("golden")) if isinstance(review, dict) else False
by_profile.setdefault(name, []).append(
{"pw": pw, "dur": dur, "rs": _resample(pw, _RESAMPLE_L)}
{"ts": ts, "pw": pw, "dur": dur, "golden": golden}
)
return by_profile
def _snaps(by_profile: dict[str, list[dict]], exclude: tuple[str, int] | None) -> list[dict[str, Any]]:
def _regrid(item: dict[str, Any], dt: float, cache: dict) -> list[float] | None:
"""One cycle's curve on a ``dt``-second grid, cached per (cycle, dt)."""
key = (id(item), round(float(dt), 2))
if key in cache:
return cache[key]
seg = _longest(resample_uniform(item["ts"], item["pw"], dt_s=dt, gap_s=_GAP_S))
out = seg.power.tolist() if seg is not None and len(seg.power) >= 2 else None
cache[key] = out
return out
def _envelope_avg(
name: str, pool: list[dict[str, Any]], excluded: int | None, dt: float, cache: dict
) -> list[float] | None:
"""The pool's DTW-warped envelope average, re-gridded to the query's ``dt``.
This is what `ProfileStore.async_match_profile` scores against once a profile
has >= 2 confirmed cycles and no pinned golden cycle, i.e. the common case
(register item 347(e)).
Cheap enough to do honestly, which the deferral used to deny (item 354): an
envelope is built from traces, so it does not depend on the config being
tuned and survives the whole grid search, and leave-one-out changes only the
target's OWN profile - so the number of distinct envelopes is
``profiles + targets``, not ``folds x profiles``. Both caches below are keyed
accordingly.
"""
ekey = ("env", name, excluded)
built = cache.get(ekey)
if built is None:
raw = [
(it["ts"].tolist(), it["pw"].tolist(), float(it["dur"]))
for it in pool
]
try:
built = analysis.compute_envelope_worker(raw, _BASE_CFG["dtw_bandwidth"])
except Exception: # pylint: disable=broad-exception-caught
built = False # cache the failure; do not retry per target
cache[ekey] = built
if not built:
return None
grid, _lo, _hi, avg, _std, _target = built
if len(grid) < 2 or len(avg) != len(grid):
return None
gkey = ("envgrid", name, excluded, round(float(dt), 2))
if gkey in cache:
return cache[gkey]
seg = _longest(
resample_uniform(
np.asarray(grid, dtype=float), np.asarray(avg, dtype=float),
dt_s=dt, gap_s=_GAP_S,
)
)
out = seg.power.tolist() if seg is not None and len(seg.power) >= 2 else None
cache[gkey] = out
return out
def _snaps(
by_profile: dict[str, list[dict]],
exclude: tuple[str, int] | None,
dt: float,
cache: dict,
) -> list[dict[str, Any]]:
"""One snapshot per profile, re-gridded to the QUERY's ``dt``.
Mirrors production: ``profile_store`` resamples the current cycle with
``resample_adaptive`` and then re-grids every candidate to that same
``used_dt`` via ``_get_cached_sample_segment``, so index *i* is the same
elapsed time on both sides.
**The template follows production's three-way rule, in production's order**
(`ProfileStore.async_match_profile`), which is register item 356 closing item
347(e): a pinned golden cycle's own sharp trace, else the DTW-warped ENVELOPE
AVERAGE once the pool holds >= 2 cycles, else the single representative
sample - which is also the safety net when an envelope will not build. The
middle branch is the common case, and it is the one that used to be missing:
scoring one representative cycle where live matching scores the envelope
average can favour weights that win on a trace nothing is matched against.
**Cost, and the trap inside it.** An envelope is built from traces, so it does
not depend on the weights being tuned and survives the whole grid search plus
every holdout call; and leave-one-out excludes exactly ONE cycle, so only the
target's own profile gets a different template while every other profile
keeps the full-pool one, shared by every target. Distinct templates are
therefore `profiles + targets`, not `folds x profiles` - measured on the worst
real export in `cycle_data/` (12 profiles with >= 2 cycles, 63 targets), 75
templates at 79 ms a DTW warp, ~5.9 s one-off. **That figure only holds if the
cache outlives one `_top1` call.** It did not at first: `_top1` built a fresh
cache and is called ~59 times (1 base + 48 grid + 10 holdout), which measured
16.1 s -> 360.6 s on a real export. Hence the `cache` argument threaded
through the whole run - nothing in it depends on `cfg`. Do not re-scope it.
The cheap approximation (averaging the pool's regridded curves without the
DTW warp) is measured WORSE and is not what `_envelope_avg` does.
The exposure is bounded regardless: `tune_matching_config` can
only move the bounded scoring weights, never structural matching behaviour,
`revert_matching_config` undoes it, and promotion is gated on **held-out
top-1 accuracy**: `tune_matching_config` promotes only when the tuned config
beats the baseline by `margin` on at least `min_wins` of `n_splits` held-out
subsamples (4 of 5) AND the mean held-out top-1 gain is itself >= `margin`.
Not AUC - no AUC is computed anywhere in this module. The AUC gate in
CLAUDE.md is `ML_TRAINING_AUC_MARGIN`, which governs the ml/ CLASSIFIERS in
`training_task.py` and has nothing to do with the matcher's scoring weights.
"""
snaps = []
for name, items in by_profile.items():
curves, durs = [], []
for idx, it in enumerate(items):
if exclude is not None and (name, idx) == exclude:
continue
curves.append(it["rs"])
durs.append(it["dur"])
if curves:
snaps.append({
"name": name,
"avg_duration": float(np.mean(durs)),
"sample_power": np.mean(np.array(curves), axis=0).tolist(),
})
excluded = exclude[1] if (exclude is not None and exclude[0] == name) else None
pool = [
it for idx, it in enumerate(items)
if exclude is None or (name, idx) != exclude
]
if not pool:
continue
durs = [it["dur"] for it in pool]
avg = float(np.mean(durs))
# Production's three-way template rule, in production's order
# (`ProfileStore.async_match_profile`). Getting this wrong is register
# item 347(e): scoring one representative cycle where live matching
# scores the envelope average can favour weights that lose on the curve
# that actually does the matching.
curve = None
golden = [it for it in pool if it.get("golden")]
if golden:
# 1. a pinned golden cycle keeps its own sharp trace; the average
# smears the wash-phase peaks, which is why production prefers it.
curve = _regrid(golden[0], dt, cache)
elif len(pool) >= 2:
# 2. the common case: the DTW-warped envelope average.
curve = _envelope_avg(name, pool, excluded, dt, cache)
if not curve:
# 3. the single-sample fallback, and the safety net for a profile
# whose envelope will not build (too few usable points, a trace
# that resamples to nothing). Representative rather than
# arbitrary: closest to the pool's mean duration.
rep = min(pool, key=lambda it: abs(it["dur"] - avg))
curve = _regrid(rep, dt, cache)
if not curve:
continue
snaps.append({
"name": name,
"avg_duration": avg,
"sample_power": curve,
})
return snaps
def _top1(by_profile: dict[str, list[dict]], targets: list[tuple[str, int]], cfg: dict[str, Any]) -> float:
def _top1(
by_profile: dict[str, list[dict]],
targets: list[tuple[str, int]],
cfg: dict[str, Any],
cache: dict | None = None,
) -> float:
"""Fraction of the given (profile, idx) targets whose true profile ranks #1
under leave-one-out matching with the given config."""
under leave-one-out matching with the given config.
``cache`` is shared ACROSS calls on purpose. Nothing in it depends on
``cfg``: the re-gridded curves are keyed by (cycle, dt) and the envelopes by
(profile, excluded index), while ``cfg`` only changes the scoring weights.
Left per-call - as it was - the grid search rebuilds every envelope for each
of its ~59 configs, which measured 16 s -> 361 s on one real export. This is
what makes the `profiles + targets` cost in register item 354 real rather
than theoretical."""
if not targets:
return 0.0
correct = 0
total = 0
if cache is None:
cache = {}
for name, idx in targets:
it = by_profile[name][idx]
snaps = _snaps(by_profile, exclude=(name, idx))
# The query defines the grid, exactly as in production.
segments, used_dt = resample_adaptive(
it["ts"], it["pw"], min_dt=_MIN_DT, gap_s=_GAP_S
)
seg = _longest(segments)
if seg is None or len(seg.power) < 4:
continue
snaps = _snaps(by_profile, (name, idx), used_dt, cache)
if len(snaps) < 2:
continue
cands = analysis.compute_matches_worker(it["pw"], it["dur"], snaps, cfg)
cands = analysis.compute_matches_worker(
seg.power.tolist(), it["dur"], snaps, cfg
)
total += 1
if cands and cands[0]["name"] == name:
correct += 1
return correct / total if total else 0.0
_BASE_CFG = {"min_duration_ratio": 0.10, "max_duration_ratio": 1.5}
# The full production matcher config. A PARTIAL config does not inherit the
# production defaults - `compute_matches_worker` has its own fallbacks - so
# omitting a key here would tune the weights against a pipeline that never
# ships. `energy_mode` is added per device type at the call site.
_BASE_CFG = {
"min_duration_ratio": DEFAULT_PROFILE_MATCH_MIN_DURATION_RATIO,
"max_duration_ratio": DEFAULT_PROFILE_MATCH_MAX_DURATION_RATIO,
"dtw_bandwidth": DEFAULT_DTW_BANDWIDTH,
"dtw_mode": DEFAULT_DTW_MODE,
}
#: Bounded scoring weights the tuner may promote. All live in [0, 1], so a tuned
#: config can only shift emphasis (shape vs level vs energy, and how much the DTW
@@ -211,10 +400,13 @@ def tune_matching_config(
# so promoted weights are consistent with the live matcher.
base = {**_BASE_CFG, "energy_mode": analysis.stage4_energy_mode(device_type)}
# Candidate: the grid config with the best top-1 on the SEARCH pool only.
best_search = _top1(by_profile, search_pool, base)
# One cache for the entire run: the grid search and both holdout arms all
# reuse the same envelopes and re-gridded curves (see `_top1`).
shared: dict = {}
best_search = _top1(by_profile, search_pool, base, shared)
best_cfg = base
for extra in _grid():
acc = _top1(by_profile, search_pool, {**base, **extra})
acc = _top1(by_profile, search_pool, {**base, **extra}, shared)
if acc > best_search:
best_search, best_cfg = acc, {**base, **extra}
override = {k: best_cfg[k] for k in OVERRIDE_KEYS if k in best_cfg}
@@ -231,8 +423,8 @@ def tune_matching_config(
pool = list(holdout_pool)
r.shuffle(pool)
held = pool[: max(1, len(pool) // 2)]
bt = _top1(by_profile, held, base)
tt = _top1(by_profile, held, best_cfg)
bt = _top1(by_profile, held, base, shared)
tt = _top1(by_profile, held, best_cfg, shared)
base_tests.append(bt)
tuned_tests.append(tt)
if tt - bt >= margin:
@@ -246,6 +438,21 @@ def tune_matching_config(
if promoted:
reason = f"beat baseline on {wins}/{n_splits} held-out subsamples"
elif not has_override:
# AUDITED and correct (register item 356), because it fires on every real
# export in `cycle_data/` and that looks like a stuck mechanism. It is
# not. `_BASE_CFG` deliberately omits the four OVERRIDE_KEYS, so
# `compute_matches_worker` falls back to `MATCH_CORR_WEIGHT` /
# `MATCH_DURATION_WEIGHT` / `MATCH_ENERGY_WEIGHT` /
# `MATCH_DTW_ENSEMBLE_W` - the tuner's baseline IS production's default,
# not a partial config with different fallbacks. `override` is therefore
# empty exactly when no grid entry beat that baseline, which is what this
# string says. The weights do reach the scorer: on the least saturated
# real device (12 profiles, 63 targets, base top-1 0.635) the grid
# produces three distinct scores, all <= base. The grid also contains the
# exact default combination (0.45 / 0.22 / 0.22 / 0.70), so the defaults
# are evaluated on equal terms, and `acc > best_search` is strict so a tie
# keeps them. Consistent with the corpus result that matcher top-1 is
# near-saturated: there is nothing here for per-device tuning to win.
reason = "defaults already optimal (no override)"
elif not enough_wins:
reason = f"only {wins}/{n_splits} held-out subsamples beat baseline by margin"