Updated apps

This commit is contained in:
2026-07-20 22:52:35 -04:00
parent 28a8cb98f6
commit a0c3271743
1164 changed files with 94781 additions and 6892 deletions
+699 -64
View File
@@ -50,6 +50,8 @@ from homeassistant.requirements import (
pip_kwargs,
)
from homeassistant.util.package import install_package
from packaging.requirements import InvalidRequirement, Requirement
from packaging.utils import canonicalize_name
from packaging.version import InvalidVersion, Version
from .const import (
@@ -83,6 +85,8 @@ from .const import (
)
if TYPE_CHECKING:
from collections.abc import Callable
from homeassistant.config_entries import ConfigEntry
_LOGGER = logging.getLogger(__name__)
@@ -92,13 +96,20 @@ _LOGGER = logging.getLogger(__name__)
# from the same refresh token on every start regardless.
_ACCESS_TOKEN_TTL = timedelta(days=3650)
# Readiness probe: how long to wait for the server thread to accept a loopback
# TCP connection before declaring the start failed. Generous on purpose: a
# cold import of the fastmcp tree takes 1-3s on real hardware but has been
# observed to exceed 30s on QEMU-emulated HAOS (the e2e lane), and a single
# readiness timeout fails the bring-up outright — there is no retry. Real
# deployments only pay this budget on the failure path.
_READY_TIMEOUT_SECONDS = 90.0
# Readiness probe: fail the bring-up only when there is no observable startup
# progress (no new modules landing in sys.modules, no phase advance) for this
# long. The previous flat 90s deadline assumed real hardware imports the
# server tree in seconds — issue #1904 (HA Green) showed a cold import alone
# can take minutes there, and killing the still-importing worker at the
# deadline is what created the orphaned-thread/port-collision cascade: the
# worker cannot be joined mid-import, lingered as a zombie, and later bound
# the port out from under the retry. The module count is process-wide (see
# _progress_signature), so a wedged worker trips the stall budget on a quiet
# instance and the absolute cap below at the latest.
_READY_STALL_TIMEOUT_SECONDS = 90.0
# Absolute ceiling on one bring-up regardless of apparent progress, matching
# the HAOS e2e lane's own 600s readiness deadline.
_READY_TOTAL_CAP_SECONDS = 600.0
_READY_POLL_INTERVAL_SECONDS = 0.5
# How long to wait for the worker thread to exit on stop before giving up and
@@ -113,12 +124,66 @@ _PIP_INSTALL_TIMEOUT_SECONDS = 300
# subprocess can never tie up an executor thread indefinitely.
_PIP_UNINSTALL_TIMEOUT_SECONDS = 120
# How long a bring-up waits for an install job orphaned by a CANCELLED
# previous bring-up before giving up: asyncio cancellation detaches the
# awaiter but an executor pip job runs to completion regardless. Sized to
# the same absolute budget as the readiness cap: pip's own timeout (300s)
# is per-download, so a healthy cold install pulling the whole dependency
# tree on slow hardware can legitimately run for minutes — a tighter bound
# would misreport it as stuck. On expiry the bring-up fails with a clear
# message and the next reload retries.
_PENDING_INSTALL_WAIT_SECONDS = 600.0
# The in-process connection API was added with the embedded server in 7.10.0.
# Older distributions can be left behind by an unsupported Core version's
# constraints and must never enter the worker thread.
MIN_EMBEDDED_SERVER_VERSION = "7.10.0"
def _derive_loopback_url(hass: HomeAssistant) -> tuple[str, bool | None]:
"""Resolve the loopback base URL for HA core from the http integration.
Returns ``(url, verify_ssl)`` where ``verify_ssl`` is ``False`` when the
URL is ``https`` (HA's certificate is issued for its hostname, never for
127.0.0.1, so verification on the loopback hop can only fail) and ``None``
when no override of the server's default is needed.
The hardcoded ``http://127.0.0.1:8123`` default this replaces broke every
instance with ``http.ssl_certificate`` configured (issue #1890): port 8123
speaks TLS there, so the server's plaintext REST/WS calls died with
"Server disconnected without sending a response" / "did not receive a
valid HTTP response" on every tool call — while the MCP handshake and
tools/list (no HA round-trip) kept working. A custom ``server_port``
similarly broke the hardcoded port. Both live in ``hass.config.api``,
set by the ``http`` integration this component depends on; the constant
remains the fallback if it is ever absent.
"""
api = getattr(hass.config, "api", None)
if api is None:
# Leave a trail: if this ever fires on a real instance, the resulting
# failure looks exactly like issue #1890 (TLS loopback broken, MCP
# handshake fine) and took a live reproduction to diagnose last time.
_LOGGER.debug(
"hass.config.api unavailable; using hardcoded loopback default %s",
DEFAULT_LOOPBACK_URL,
)
return DEFAULT_LOOPBACK_URL, None
# Strict type checks (not coercion / truthiness): a malformed api object
# must resolve to the plaintext default on port 8123, never to a surprise
# port or https flip. bool is excluded because it is an int subclass.
port_raw = getattr(api, "port", None)
port = (
port_raw
if isinstance(port_raw, int)
and not isinstance(port_raw, bool)
and 0 < port_raw <= 65535
else 8123
)
if getattr(api, "use_ssl", False) is True:
return f"https://127.0.0.1:{port}", False
return f"http://127.0.0.1:{port}", None
class EmbeddedServerError(Exception):
"""Raised when the in-process ha-mcp server could not be installed or started.
@@ -146,9 +211,17 @@ class EmbeddedServerManager:
options = entry.options
self._port: int = int(options.get(OPT_SERVER_PORT, DEFAULT_SERVER_PORT))
self._bind_host: str = str(options.get(OPT_BIND_HOST, DEFAULT_BIND_HOST))
self._server_url: str = str(
options.get(OPT_SERVER_URL) or DEFAULT_LOOPBACK_URL
).rstrip("/")
# An explicit server_url override wins verbatim (the operator manages
# scheme/verification themselves via the settings UI). A stored value
# equal to DEFAULT_LOOPBACK_URL is treated as no-override: the options
# form used to pre-fill that constant as suggested_value, so existing
# entries carry it without the user ever having chosen it.
_url_override = str(options.get(OPT_SERVER_URL) or "").rstrip("/")
self._loopback_verify_ssl: bool | None = None
if _url_override and _url_override != DEFAULT_LOOPBACK_URL:
self._server_url: str = _url_override
else:
self._server_url, self._loopback_verify_ssl = _derive_loopback_url(hass)
self._channel: str = str(options.get(OPT_CHANNEL) or DEFAULT_CHANNEL)
# An explicit pip-spec override (the pre-release test channel) wins over
# the channel selector. DEFAULT_PIP_SPEC in the field means "no override,
@@ -189,6 +262,10 @@ class EmbeddedServerManager:
# Compared against the installed distribution after start to detect a
# stale-code worker (see _purge_ha_mcp_modules).
self._running_version: str | None = None
# Startup phase marker (plain attribute writes: init markers from the
# main thread, _note_startup_phase transitions from the worker). Read
# by the readiness poll for progress detection and error messages.
self._startup_phase: str = "not started"
@property
def port(self) -> int:
@@ -233,45 +310,39 @@ class EmbeddedServerManager:
kind="package",
)
await self._async_ensure_package()
# Read the importer registry BEFORE the package step too: replacing
# the distribution's files on disk under a live importer corrupts it
# exactly like the sys.modules purge does (review finding), so a
# mutating install is deferred while any registered worker is alive.
ready_version = await self._async_ensure_package(
defer_mutations=_prune_and_check_importing_workers()
)
access_token = await self._async_provision_token()
await self._hass.async_add_executor_job(self._prepare_config_dir)
# Drop cached ha_mcp modules so the worker imports the code that is on
# disk NOW. Without this, a reload after a pip install keeps serving
# the OLD code forever: all workers are threads of the one HA core
# process, and Python resolves ``import ha_mcp`` from sys.modules —
# installs only took effect after a full HA core restart (issue
# observed live: options saves reinstalled the package, the web UI
# footer showed the new on-disk version, yet the serving worker kept
# reporting the version it was first imported with).
#
# SKIPPED while an orphaned worker may still be importing: ripping
# entries out of sys.modules under a live importer corrupts its
# import in progress (seen on QEMU-slow HAOS, where a cold import
# can outlive both the readiness timeout and the stop-join budget).
# The post-start staleness check below surfaces the consequence
# (old code possibly serving) instead.
orphan = self._orphaned_thread
if orphan is not None and not orphan.is_alive():
self._orphaned_thread = orphan = None
if orphan is None:
_purge_ha_mcp_modules()
else:
_LOGGER.warning(
"Skipping the ha_mcp module purge: a previous worker thread "
"is still shutting down. The new worker may serve the "
"previously imported code until Home Assistant restarts."
)
self._maybe_purge_stale_modules(ready_version)
self._thread_exc = None
self._startup_phase = "waiting for the worker thread"
self._thread = threading.Thread(
target=self._thread_main,
args=(access_token,),
name="ha-mcp-server",
daemon=True,
)
self._thread.start()
# Registered by THIS (main) thread before start(): every bring-up
# runs its gate → register → start section synchronously on the one
# event loop, so another bring-up's purge gate can never interleave
# with a worker that exists but has not yet registered itself
# (review finding). The worker itself only ever deregisters.
with _IMPORTING_WORKERS_LOCK:
_IMPORTING_WORKERS.add(self._thread)
try:
self._thread.start()
except BaseException:
with _IMPORTING_WORKERS_LOCK:
_IMPORTING_WORKERS.discard(self._thread)
raise
await self._async_wait_until_ready()
@@ -402,9 +473,171 @@ class EmbeddedServerManager:
return None
return DIST_NAME_STABLE if self._channel == CHANNEL_DEV else DIST_NAME_DEV
async def _async_ensure_package(self) -> None:
def _maybe_purge_stale_modules(self, ready_version: str | None) -> None:
"""Purge cached ha_mcp modules unless doing so is unsafe or pointless.
The purge makes the next worker import the code that is on disk NOW.
Without it, a reload after a pip install keeps serving the OLD code
forever: all workers are threads of the one HA core process, and
Python resolves ``import ha_mcp`` from sys.modules — installs only
took effect after a full HA core restart (observed live: options
saves reinstalled the package, the web UI footer showed the new
on-disk version, yet the serving worker kept reporting the version
it was first imported with).
SKIPPED while any previous worker may still be importing — ripping
entries out of sys.modules under a live importer corrupts its import
in progress. Two guards cover that: the per-manager orphan (a worker
this manager's stop could not join), and the process-global
``_IMPORTING_WORKERS`` registry, because every bring-up constructs a
FRESH manager (async_bring_up_server) and an entry reload during a
slow cold import otherwise hands the purge to a manager that has
never heard of the still-importing worker — which then crashed
mid-import with KeyError: 'ha_mcp.config' (issue #1904, live on
1.1.1-dev.107). The post-start staleness check surfaces the
consequence (old code possibly serving) instead.
Also skipped when the cached modules already ARE the generation on
disk: purging on every attempt made each retry pay the full cold
import again, so slow hardware that missed the readiness window once
could never recover (#1904). Never skipped under a pip-spec override
— the one workflow where a reinstall can change the code without
changing the version string (re-pointed tarball/pin), which a
version-keyed skip would serve stale; channel installs mint a
distinct version per build.
"""
orphan = self._orphaned_thread
if orphan is not None and not orphan.is_alive():
self._orphaned_thread = orphan = None
importing_busy = _prune_and_check_importing_workers()
if orphan is not None:
_LOGGER.warning(
"Skipping the ha_mcp module purge: a previous worker thread "
"is still shutting down. The new worker may serve the "
"previously imported code until Home Assistant restarts."
)
elif importing_busy:
# Distinct from the orphan message: that worker is still STARTING
# (mid cold-import, the #1904 incident shape), not shutting down —
# naming the actual state matters when reading logs during one.
_LOGGER.warning(
"Skipping the ha_mcp module purge: a previous bring-up's "
"worker thread is still importing. The new worker may serve "
"the previously imported code until Home Assistant restarts."
)
elif (
not self._pip_spec_override
and ready_version is not None
and _CACHED_IMPORT_VERSION is not None
and ready_version == _CACHED_IMPORT_VERSION
):
_LOGGER.debug(
"Cached ha_mcp modules already match installed version %s; "
"skipping the module purge (warm start)",
ready_version,
)
else:
_purge_ha_mcp_modules()
async def _async_run_tracked_install_job(
self, func: Callable[[], object]
) -> object:
"""Run a package-mutating executor job, tracked process-wide.
Registration happens BEFORE dispatch and completion is signalled on
the executor thread in a finally, so a bring-up cancelled mid-job
(install or uninstall) leaves behind a waitable job instead of an
invisible one.
"""
global _PENDING_INSTALL_DONE
done = threading.Event()
with _PENDING_INSTALL_LOCK:
_PENDING_INSTALL_DONE = done
def _run() -> object:
global _PENDING_INSTALL_DONE
try:
return func()
finally:
done.set()
with _PENDING_INSTALL_LOCK:
# A newer job may already have replaced the slot.
if _PENDING_INSTALL_DONE is done:
_PENDING_INSTALL_DONE = None
try:
# Shielded: cancelling the awaiter must NOT cancel the executor
# job. Unshielded, a cancel landing while the job is still
# QUEUED removes it from the pool — _run never starts, nothing
# ever sets the event, and the next bring-up waits the full
# budget on a job that does not exist (review finding). With the
# shield, _run always runs and its finally always fires; the
# awaiter still detaches immediately on cancel.
return await asyncio.shield(self._hass.async_add_executor_job(_run))
except asyncio.CancelledError:
# The job is queued or running and its finally will clear the
# slot; leaving it registered is the whole point — the next
# bring-up must wait it out.
raise
except BaseException:
# A DISPATCH failure (e.g. executor already shut down) means
# _run never ran and nothing will ever set the event — clear our
# own registration or the next bring-up waits the full budget on
# a job that does not exist. The identity check makes this a
# no-op if _run did run and already cleaned up.
with _PENDING_INSTALL_LOCK:
if _PENDING_INSTALL_DONE is done:
_PENDING_INSTALL_DONE = None
raise
async def _async_wait_for_pending_install(self) -> None:
"""Wait out an install job orphaned by a cancelled previous bring-up.
Raises :class:`EmbeddedServerError` if it is still running after the
bounded wait — mutating the package (or importing from it) underneath
a live pip job is the same corruption class as purging sys.modules
under a live importer.
The wait occupies one pooled executor thread (bounded): the setter
runs on an executor thread with no handle to this loop, so a
loop-side wakeup would need cross-thread plumbing this rare recovery
path does not justify.
"""
with _PENDING_INSTALL_LOCK:
pending = _PENDING_INSTALL_DONE
if pending is None or pending.is_set():
return
_LOGGER.warning(
"A previous bring-up's install job is still running on the "
"executor (the bring-up was cancelled but pip cannot be); "
"waiting for it to finish before touching the ha-mcp package."
)
finished = await self._hass.async_add_executor_job(
pending.wait, _PENDING_INSTALL_WAIT_SECONDS
)
if not finished:
raise EmbeddedServerError(
f"A previous install job was still running after "
f"{_PENDING_INSTALL_WAIT_SECONDS:.0f}s; refusing to modify "
"the ha-mcp package underneath it. Reload the integration "
"to retry.",
kind="package",
)
async def _async_ensure_package(
self, *, defer_mutations: bool = False
) -> str | None:
"""Ensure ``ha-mcp`` is importable, installing the pip spec if needed.
Returns the installed version that the worker is about to run, for the
caller's warm-cache purge decision.
``defer_mutations=True`` (a previous bring-up's worker is still
importing) downgrades any would-be uninstall/force-install to the
non-mutating fast path: replacing the distribution's files on disk
under a live importer corrupts it the same way a sys.modules purge
does. The deferred update applies on the next reload or HA restart.
With auto-update on (the default) both channels install their
distribution UNPINNED, so every entry reload / HA restart must pick up
the newest build. Such a spec ALWAYS takes the force-install path
@@ -419,9 +652,12 @@ class EmbeddedServerManager:
When that spec matches the one last installed and the package imports,
delegate the "already satisfied?" decision to Home Assistant's
requirements manager; a pinned spec does not move, so there is nothing to
upgrade to. A CHANGED spec (a new override, a toggled auto-update, a
channel switch) still falls through to the force-install path below so
the change actually takes effect.
upgrade to. A CHANGED spec (a new override, a cleared override, a
toggled auto-update, a channel switch) falls through to the
force-install path below — and additionally uninstalls the replaced
distribution first (:meth:`_async_remove_replaced_source`), because
``upgrade=True`` alone decides by version and a changed SOURCE can keep
the version string (issue #1914).
On a channel switch the other channel's distribution is uninstalled first
(:meth:`_async_remove_conflicting_dist`): ``ha-mcp`` and ``ha-mcp-dev``
@@ -437,14 +673,14 @@ class EmbeddedServerManager:
Never imports ``ha_mcp`` in this (main) process — that happens only inside
the worker thread.
"""
await self._async_wait_for_pending_install()
stored_spec = self._entry.data.get(DATA_LAST_PIP_SPEC)
installed_version = await self._hass.async_add_executor_job(
_installed_ha_mcp_version
)
pending_version = str(
self._entry.data.get(DATA_PENDING_INSTALL_VERSION) or ""
).strip()
pending_version = self._pending_install_version(defer_mutations)
target_dist = dist_for_channel(self._channel)
if not self._pip_spec_override and pending_version:
# Pin to the requested version. Its own value differs from
@@ -498,13 +734,19 @@ class EmbeddedServerManager:
and installed_version is not None
and _is_compatible_embedded_version(installed_version)
)
deferred = False
if fast_path_ok:
await self._async_process_requirements_fast()
elif defer_mutations:
deferred = True
await self._async_defer_package_mutations(installed_version)
else:
await self._async_remove_conflicting_dist()
await self._async_remove_legacy_target(target_dist, installed_version)
await self._async_remove_replaced_source(stored_spec, installed_version)
await self._async_force_install()
version: str | None
if not self._pip_spec_override and self._channel == CHANNEL_DEV:
version = await self._hass.async_add_executor_job(
_installed_ha_mcp_version, target_dist
@@ -527,8 +769,13 @@ class EmbeddedServerManager:
kind="package",
)
_LOGGER.info("HA-MCP in-process server package ready (version %s)", version)
if stored_spec != self._pip_spec:
# A DEFERRED spec change must not be recorded as installed: with the
# stored spec advanced, the next reload would see "unchanged", take the
# fast path (for a stable spec) and skip the replaced-source uninstall,
# so the deferred change would silently never apply.
if not deferred and stored_spec != self._pip_spec:
self._store_installed_spec()
return version
async def _async_remove_legacy_target(
self, target_dist: str, installed_version: str | None
@@ -555,6 +802,165 @@ class EmbeddedServerManager:
)
await self._async_remove_distribution(target_dist)
def _pending_install_version(self, defer_mutations: bool) -> str:
"""Return the update entity's pending-install version, or ``""``.
Always empty while mutations are deferred, leaving the marker in
``entry.data`` untouched: the deferred branch runs no install, and
consuming the marker without an attempt would lose the user's Install
click entirely (with auto-update off, the next reload re-pins to the
OLD installed version). One-shot means one real ATTEMPT — a deferred
bring-up never attempts, so the marker survives to the next
undeferred reload (review finding on #1923).
"""
if defer_mutations:
return ""
return str(self._entry.data.get(DATA_PENDING_INSTALL_VERSION) or "").strip()
async def _async_defer_package_mutations(
self, installed_version: str | None
) -> None:
"""Handle the defer-mutations branch of :meth:`_async_ensure_package`.
A previous bring-up's worker is still importing, so the package files
must not be replaced under it. With an importable build on disk
(``installed_version`` non-None — importability only, compatibility is
checked by the caller afterwards) nothing is touched at all — not even
the requirements manager, which installs any unsatisfied spec and
would mutate exactly like the deferred force install. When nothing
imports there are no distribution files to replace under the live
importer, and without an install this bring-up cannot produce a
server at all, so the requirements manager still runs.
"""
_LOGGER.warning(
"Deferring the ha-mcp install/upgrade: a previous bring-up's "
"worker thread is still importing, and replacing the package "
"files under it could corrupt that import. The currently "
"installed build will be used; reload the integration (or "
"restart Home Assistant) to apply the update."
)
if installed_version is None:
await self._async_process_requirements_fast()
def _replaced_dist_name(self) -> str | None:
"""Return the distribution whose presence could no-op the new spec.
This is the distribution the effective spec installs *by name* — the
channel's distribution for a channel spec, or the named distribution
of an override that parses as a requirement (a pin like
``ha-mcp==X``, matched against the two known channel dists). It is
deliberately NOT the channel's dist for every override: a repo
tarball installs as ``ha-mcp`` regardless of the selected channel, so
keying on the channel would miss the dev-channel + override case.
Returns None for an override that names an unknown distribution or
does not parse as a requirement at all (a direct URL): the installer
re-fetches and rebuilds URL requirements under ``upgrade=True``
regardless of the installed version, so a URL install is already
real and nothing needs removing.
"""
if not self._pip_spec_override:
return dist_for_channel(self._channel)
try:
name = canonicalize_name(Requirement(self._pip_spec_override).name)
except InvalidRequirement:
return None
for dist_name in (DIST_NAME_STABLE, DIST_NAME_DEV):
if name == canonicalize_name(dist_name):
return dist_name
return None
async def _async_remove_replaced_source(
self, stored_spec: str | None, installed_version: str | None
) -> None:
"""Uninstall the replaced distribution when the requested source changed.
The forced install that follows relies on ``upgrade=True``, and the
installer decides "already satisfied" by VERSION alone — but a source
change can keep the version string. A PR branch's committed
``project.version`` equals the release it branched from (only release
automation bumps it), so its tarball installs with that same version
string; clearing the override then resolves the channel spec to the
exact version already on disk and the install swaps nothing, leaving
the PR code running while the entry reports a clean channel install
(issue #1914). The same version-blindness bites a manual spec edit
that pins the version already installed. The installer cannot see the
difference, so when the spec that produced the current install
differs from the one about to be installed, the distribution the new
spec resolves to by name (:meth:`_replaced_dist_name`) is removed
first — the install that follows is then unconditionally real.
Skipped when nothing is installed, when the last-installed spec is
unknown (nothing to compare: first install, or entry data predating
the spec tracking), when the spec is unchanged (the routine
reload/restart path, where ``upgrade=True`` alone is correct and an
uninstall would churn — and briefly break — a healthy install on
every restart), when the new spec is a direct URL (always installs
for real), when the named distribution is not installed (e.g. a
cross-channel switch already removed it), when the stored spec is
an index requirement on the SAME distribution (a repin — e.g.
toggling auto-update rewrites bare ``ha-mcp`` to ``ha-mcp==X`` —
draws from the same index either way, so version resolution is
faithful and uninstalling a healthy install on a preference toggle
would only add an offline-breakage window), or when the new spec is
an exact pin on a version provably different from the installed one
(the install cannot no-op, so the working build stays in place as
the fallback if it fails).
Unlike the other pre-install uninstalls this one is NOT best-effort:
if the distribution survives a failed uninstall, the forced install
would no-op as "already satisfied", the new spec would be persisted,
and the next reload would see it as unchanged — reproducing #1914 and
then permanently masking it. Raising instead keeps the stored spec on
the old value, so the next reload retries the whole source change.
"""
if installed_version is None or stored_spec is None:
return
if stored_spec == self._pip_spec:
return
replaced_dist = self._replaced_dist_name()
if replaced_dist is None:
return
if _spec_is_index_requirement_on(stored_spec, replaced_dist):
# Same distribution, same index — only the pin changed. The old
# code on disk came from the index too, so "already satisfied by
# version" is the truth, not the #1914 lie.
return
pinned = _exact_pinned_version(self._pip_spec)
if pinned is not None:
try:
version_moves = Version(pinned) != Version(installed_version)
except InvalidVersion:
version_moves = False # unprovable — keep the uninstall
if version_moves:
# The new pin cannot be satisfied by the installed version, so
# the forced install is guaranteed to be real without any
# uninstall — and keeping the working build in place preserves
# it as the fallback if that install fails (e.g. offline).
return
if not await self._hass.async_add_executor_job(_dist_installed, replaced_dist):
return
_LOGGER.info(
"The requested server source changed (%r -> %r); removing the "
"installed %r first so the reinstall cannot be skipped as "
"already satisfied",
stored_spec,
self._pip_spec,
replaced_dist,
)
removed = await self._async_remove_distribution(replaced_dist)
if not removed and await self._hass.async_add_executor_job(
_dist_installed, replaced_dist
):
raise EmbeddedServerError(
f"Could not remove the installed {replaced_dist!r} (from "
f"{stored_spec!r}) before installing {self._pip_spec!r}: the "
"installer would report the new source as already satisfied "
"and keep the old code running. Uninstall details are logged "
"above; reload the integration to retry the source change.",
kind="package",
)
async def _async_process_requirements_fast(self) -> None:
"""Fast path: let HA's requirements manager satisfy the override spec."""
try:
@@ -583,7 +989,7 @@ class EmbeddedServerManager:
kwargs["timeout"] = max(
int(kwargs.get("timeout") or 0), _PIP_INSTALL_TIMEOUT_SECONDS
)
installed = await self._hass.async_add_executor_job(
installed = await self._async_run_tracked_install_job(
partial(install_package, self._pip_spec, upgrade=True, **kwargs)
)
if not installed:
@@ -624,15 +1030,25 @@ class EmbeddedServerManager:
)
await self._async_remove_distribution(other)
async def _async_remove_distribution(self, dist_name: str) -> None:
"""Remove a distribution from the same target used for installation."""
async def _async_remove_distribution(self, dist_name: str) -> bool:
"""Remove a distribution from the same target used for installation.
Tracked like the install: an uninstall mutates the same package files.
Returns whether the uninstall subprocess reported success; callers
decide whether a failure is best-effort (channel-conflict / legacy
cleanup) or fatal (the replaced-source removal, whose failure would
silently void the reinstall — see ``_async_remove_replaced_source``).
"""
target = pip_kwargs(self._hass.config.config_dir).get("target")
if target is None:
await self._hass.async_add_executor_job(_uninstall_distribution, dist_name)
result = await self._async_run_tracked_install_job(
partial(_uninstall_distribution, dist_name)
)
else:
await self._hass.async_add_executor_job(
result = await self._async_run_tracked_install_job(
partial(_uninstall_distribution, dist_name, target=target)
)
return bool(result)
def _store_installed_spec(self) -> None:
"""Persist the pip spec just installed so a restart skips the reinstall."""
@@ -724,6 +1140,15 @@ class EmbeddedServerManager:
# -- worker thread -----------------------------------------------------
def _note_startup_phase(self, phase: str) -> None:
"""Publish the worker's startup phase (a plain attribute write).
Read by the readiness poll: a phase advance counts as progress, and the
failure message names the phase the worker was last seen in.
"""
self._startup_phase = phase
_LOGGER.debug("HA-MCP in-process server startup: %s", phase)
def _thread_main(self, access_token: str) -> None:
"""Thread entry point: stage non-secret env, then run the server.
@@ -744,12 +1169,38 @@ class EmbeddedServerManager:
stop_event = asyncio.Event()
self._loop = loop
self._stop_event = stop_event
# Registration in _IMPORTING_WORKERS happened on the MAIN thread,
# before start() — see async_start. This thread only deregisters:
# in _serve once the import section completes, and in the finally
# below on exit as the backstop.
try:
loop.run_until_complete(self._serve(access_token, stop_event))
except SystemExit as err:
# uvicorn signals a startup failure (e.g. the port is already in
# use) with SystemExit(STARTUP_FAILURE), which ``except Exception``
# misses — live issue #1904 saw the real bind error surface only
# in HA's generic task-exception log while the component reported
# a bare readiness timeout. Unwrap the original error so the
# repair issue names the actual cause; a bare SystemExit (no
# chained exception) is reported by repr so an empty/zero exit
# code still reads as what it is. The phase names where in
# _serve the exit happened instead of hardcoding a bind failure.
cause = err.__context__ or err.__cause__
detail = str(cause) if cause is not None else repr(err)
self._thread_exc = EmbeddedServerError(
f"the server exited during startup ({self._startup_phase}): {detail}"
)
_LOGGER.error(
"HA-MCP in-process server exited during startup (%s): %s",
self._startup_phase,
detail,
)
except Exception as err:
self._thread_exc = err
_LOGGER.exception("HA-MCP in-process server thread crashed")
finally:
with _IMPORTING_WORKERS_LOCK:
_IMPORTING_WORKERS.discard(threading.current_thread())
# Teardown is best-effort but never SILENT (review finding): a
# raise here must not mask the primary outcome, yet a recurring
# cleanup failure (leaking executor threads across reloads) has
@@ -779,6 +1230,7 @@ class EmbeddedServerManager:
# Hand ha-mcp the loopback URL + provisioned admin token in memory, before
# the server (and its settings singleton) is built. Keeping the token out
# of os.environ is the whole point of the in-process channel.
self._note_startup_phase("importing the server package")
import ha_mcp.config as _hamcp_config
# Record which code generation this worker imported. Prefer the
@@ -786,6 +1238,11 @@ class EmbeddedServerManager:
# ha_mcp.__version__ itself checks stable first and stale stable
# metadata can otherwise make a fresh dev worker look outdated.
self._running_version = _running_ha_mcp_version(self._channel)
# The cache in sys.modules now holds this generation — remembered
# process-wide so the next start can skip the purge when the install
# has not changed (issue #1904).
global _CACHED_IMPORT_VERSION
_CACHED_IMPORT_VERSION = self._running_version
# Drop any settings singleton cached by a PREVIOUS start in this same
# Python process: an entry reload must re-read the override files
@@ -805,12 +1262,33 @@ class EmbeddedServerManager:
"entry may serve stale override values until HA restarts"
)
_hamcp_config.set_embedded_connection(self._server_url, access_token)
if self._loopback_verify_ssl is None:
_hamcp_config.set_embedded_connection(self._server_url, access_token)
else:
try:
_hamcp_config.set_embedded_connection(
self._server_url,
access_token,
verify_ssl=self._loopback_verify_ssl,
)
except TypeError:
# Server predates the verify_ssl parameter (< the release
# carrying issue #1890's fix). Register url+token the old way;
# on an SSL-enabled instance the wss loopback will fail
# certificate verification until the server package updates —
# no worse than the plaintext failure it replaces.
_LOGGER.warning(
"Installed ha-mcp server does not accept verify_ssl for "
"the embedded connection; loopback TLS verification stays "
"enabled until the server package updates"
)
_hamcp_config.set_embedded_connection(self._server_url, access_token)
# Imported here, in the worker thread, after the connection is registered.
from ha_mcp.server import HomeAssistantSmartMCPServer
from ha_mcp.settings_ui import register_settings_routes
self._note_startup_phase("building the server")
server = HomeAssistantSmartMCPServer()
# Startup observability (no secrets): confirm the in-memory connection
@@ -849,6 +1327,7 @@ class EmbeddedServerManager:
# Parity with the CLI HTTP runner: serve the web settings UI under the
# same secret path as the MCP endpoint.
self._note_startup_phase("registering web routes")
register_settings_routes(server.mcp, server, secret_path=self._secret_path)
# Parity with the CLI HTTP runner: answer a browser GET on the MCP path
@@ -891,6 +1370,16 @@ class EmbeddedServerManager:
else:
ensure_host_origin_guard_default_off()
# The cold import — the multi-minute window that crashed in #1904 —
# is complete: every explicit ha_mcp import in _serve precedes this
# line. Leave the registry so a long-running healthy server never
# blocks later bring-ups' purges (removal from sys.modules cannot
# unload already-bound modules; only in-flight imports are
# corruptible, and any later lazy import is outside the window this
# registry protects).
with _IMPORTING_WORKERS_LOCK:
_IMPORTING_WORKERS.discard(threading.current_thread())
app = server.mcp.http_app(path=self._secret_path, stateless_http=True)
config = uvicorn.Config(
app,
@@ -905,6 +1394,7 @@ class EmbeddedServerManager:
)
uv_server = uvicorn.Server(config)
self._note_startup_phase("starting the HTTP listener")
stop_task = asyncio.create_task(stop_event.wait())
async with server.mcp._lifespan_manager():
serve_task = asyncio.create_task(uv_server.serve())
@@ -924,15 +1414,37 @@ class EmbeddedServerManager:
# Surface a server that exited on its own (bind failure, etc.).
serve_task.result()
def _progress_signature(self) -> tuple[int, str]:
"""Snapshot the observable startup progress of the worker thread.
``len(sys.modules)`` moves continuously while the worker grinds
through a cold import (the single longest startup step — minutes on
slow hardware, issue #1904), and the published phase moves between
steps. Any change in the pair counts as progress. The module count is
PROCESS-wide — an approximation: nothing finer-grained is observable
from outside a thread stuck inside one ``import`` statement, and any
other HA thread importing concurrently also refreshes the stall
budget. Erring toward patience is the point; the absolute cap bounds
the wait regardless.
"""
return (len(sys.modules), self._startup_phase)
async def _async_wait_until_ready(self) -> None:
"""Poll a loopback TCP connect until the server accepts, or fail.
On failure (timeout or an early thread crash) stops the thread and raises
Patience is progress-based: the wait only gives up when there is no
observable progress for ``_READY_STALL_TIMEOUT_SECONDS`` (or the
absolute ``_READY_TOTAL_CAP_SECONDS`` ceiling is hit). A slow cold
import keeps the wait alive; a wedged worker is caught by the stall
budget on a quiet instance, by the cap at the latest (the progress
signal is process-wide). On failure stops the thread and raises
:class:`EmbeddedServerError` so the caller leaves the webhook
unregistered and files a repair issue.
"""
deadline = self._hass.loop.time() + _READY_TIMEOUT_SECONDS
while self._hass.loop.time() < deadline:
start = self._hass.loop.time()
last_progress = start
last_signature = self._progress_signature()
while True:
if self._thread_exc is not None:
raise EmbeddedServerError(
f"HA-MCP in-process server failed to start: {self._thread_exc}"
@@ -948,15 +1460,32 @@ class EmbeddedServerManager:
self._port,
)
return
now = self._hass.loop.time()
signature = self._progress_signature()
if signature != last_signature:
last_signature = signature
last_progress = now
if now - start >= _READY_TOTAL_CAP_SECONDS:
failure = (
f"HA-MCP in-process server did not become reachable on "
f"port {self._port} within {_READY_TOTAL_CAP_SECONDS:.0f}s "
f"(last startup phase: {self._startup_phase})."
)
break
if now - last_progress >= _READY_STALL_TIMEOUT_SECONDS:
failure = (
f"HA-MCP in-process server did not become reachable on "
f"port {self._port}: no startup progress observed for "
f"{_READY_STALL_TIMEOUT_SECONDS:.0f}s (last phase: "
f"{self._startup_phase}; {now - start:.0f}s since start)."
)
break
await asyncio.sleep(_READY_POLL_INTERVAL_SECONDS)
# Timed out — tear the thread down so we never leave a half-started
# Gave up — tear the thread down so we never leave a half-started
# server behind an unregistered webhook.
await self.async_stop()
raise EmbeddedServerError(
f"HA-MCP in-process server did not become reachable on port "
f"{self._port} within {_READY_TIMEOUT_SECONDS:.0f}s."
)
raise EmbeddedServerError(failure)
async def _async_probe_port(self) -> bool:
"""Return True if a loopback TCP connection to the server port succeeds.
@@ -977,6 +1506,68 @@ class EmbeddedServerManager:
return True
# Worker threads that may still be executing their ha_mcp imports. Purging
# sys.modules while any of them is alive corrupts the in-flight import
# (KeyError from frozen importlib — issue #1904). Process-global because a
# manager is recreated on every bring-up, so per-manager orphan tracking
# cannot see a previous manager's abandoned worker. The spawning (main)
# thread registers the worker BEFORE start() — see async_start — and the
# worker deregisters once its import section completes (or on thread exit,
# whichever comes first). All access
# goes through the lock: CPython's GIL would make the individual set ops
# atomic, but the purge gate's prune-then-check is a composite read and the
# lock keeps its correctness independent of GIL scheduling arguments.
# Deliberate tradeoff: a worker wedged forever inside its import stays
# registered and blocks every later purge until an HA core restart — evicting
# a live importer on a timer would reintroduce the very corruption this
# registry prevents, and the skip warning plus the post-start staleness check
# surface the condition.
_IMPORTING_WORKERS_LOCK = threading.Lock()
_IMPORTING_WORKERS: set[threading.Thread] = set()
def _prune_and_check_importing_workers() -> bool:
"""Drop dead workers from the registry; return True if any live one remains."""
with _IMPORTING_WORKERS_LOCK:
_IMPORTING_WORKERS.difference_update(
[t for t in _IMPORTING_WORKERS if not t.is_alive()]
)
return bool(_IMPORTING_WORKERS)
# Completion event of the package-mutating install/uninstall job currently on
# the executor, if any. asyncio cancellation of a bring-up detaches the
# awaiter, but the executor job keeps running to completion — untracked, an
# orphaned pip could swap the distribution's files under the NEXT bring-up's
# install or its worker's cold import (found in review of PR #1911, the
# #1904 fixes; pre-existing). The dispatching coroutine registers the event BEFORE handing
# the job to the executor, and the executor fn sets it in a finally that
# survives cancellation; the next bring-up waits on it before mutating
# anything. Process-global for the same reason as _IMPORTING_WORKERS: a
# manager is recreated on every bring-up.
#
# Single slot BY DESIGN: at most one tracked job can exist at a time — the
# server entry is single-instance, an entry reload cancels-and-awaits the
# previous bring-up before setting up, and every dispatch site sits behind
# _async_wait_for_pending_install. A second concurrent dispatcher would
# overwrite the slot and silently lose the older live job — keep any new
# package-mutating call site behind the wait gate. The slot tracking wraps
# the DIRECT-pip sites (force install, uninstalls); the fast path goes
# through HA's requirements manager, which is behind the gate but untracked —
# it only dispatches pip when the package is missing outright, which cannot
# co-occur with a live orphaned job worth waiting on.
_PENDING_INSTALL_LOCK = threading.Lock()
_PENDING_INSTALL_DONE: threading.Event | None = None
# Version of the ha_mcp generation currently cached in sys.modules — set by
# the worker right after its first import lands, cleared by the purge. Process-wide
# (the module cache it describes is process-wide too). Lets a retry with an
# unchanged install keep the warm cache instead of paying the full cold import
# again (issue #1904).
_CACHED_IMPORT_VERSION: str | None = None
def _purge_ha_mcp_modules() -> None:
"""Drop every cached ``ha_mcp`` module so the next import loads fresh code.
@@ -984,11 +1575,15 @@ def _purge_ha_mcp_modules() -> None:
Python resolves imports from the process-wide ``sys.modules`` cache — so
after a pip install the next worker would silently reuse the OLD code
unless the cache is purged first. Safe here because ``ha_mcp`` is pure
Python and is only ever imported inside the (currently stopped) worker
thread; third-party dependencies are deliberately NOT purged (they are
shared with the rest of Home Assistant), so a dependency-version change
still needs an HA core restart.
Python and is only ever imported inside worker threads, and the caller's
gate guarantees no registered worker is mid-import when this runs (a
worker past its imports keeps its already-bound modules regardless);
third-party dependencies are deliberately NOT purged (they are shared
with the rest of Home Assistant), so a dependency-version change still
needs an HA core restart.
"""
global _CACHED_IMPORT_VERSION
_CACHED_IMPORT_VERSION = None
# Snapshot the keys: sys.modules can be mutated by concurrent imports on
# other threads mid-iteration (HA core is heavily threaded).
purged = [
@@ -1120,6 +1715,46 @@ def _uninstall_distribution(dist_name: str, *, target: str | None = None) -> boo
return True
def _exact_pinned_version(spec: str) -> str | None:
"""Return the version of an exact ``==``/``===`` single-clause pin, or None.
Anything else — URL specs, bare names, ranges, multi-clause specifiers —
returns None: only an exact pin lets the caller prove, without asking the
resolver, whether the installed version could satisfy the spec. A
wildcard pin (``==7.13.*``) is returned as-is; the caller's ``Version``
parse rejects it, which conservatively keeps the uninstall.
"""
try:
req = Requirement(spec)
except InvalidRequirement:
return None
if req.url:
return None
clauses = list(req.specifier)
if len(clauses) != 1 or clauses[0].operator not in ("==", "==="):
return None
return clauses[0].version
def _spec_is_index_requirement_on(spec: str, dist_name: str) -> bool:
"""Return whether ``spec`` is a plain index requirement on ``dist_name``.
True only for a PEP 508 requirement with no direct-URL part whose
canonical name matches — i.e. a spec that installs ``dist_name`` from the
package index (bare name or version pin). A direct URL (whether a plain
URL string, which does not parse as a requirement, or a ``name @ url``
form) returns False: its origin is not the index, so it is a genuine
source change for the replaced-source check.
"""
try:
req = Requirement(spec)
except InvalidRequirement:
return False
if req.url:
return False
return canonicalize_name(req.name) == canonicalize_name(dist_name)
def _is_compatible_embedded_version(version: str) -> bool:
"""Return whether a server distribution provides the embedded API."""
try: