diff options
Diffstat (limited to 'app/backends/servers.py')
| -rw-r--r-- | app/backends/servers.py | 194 |
1 files changed, 149 insertions, 45 deletions
diff --git a/app/backends/servers.py b/app/backends/servers.py index 912c09d..62eadb5 100644 --- a/app/backends/servers.py +++ b/app/backends/servers.py @@ -1,16 +1,25 @@ """Start and stop TTS backend servers from the TUI hub. Each backend's ``detect()`` returns a list of ``ServerSpec`` — the exact argv -(absolute binaries in the managed venv, no shell activation needed) and the -URL to probe for readiness. This module turns those specs into running -processes: ``start`` spawns the server, streams its output to -``app/logs/<name>-server.log``, records its pid, and polls the URL until it -accepts connections (model loads are slow, so the timeout is generous); -``stop`` terminates the process group the hub started. - -Everything here runs in the plain console tail after the curses TUI returns -(matching the wizards' build/pip streaming), so progress and log tails appear -normally. Pid/log files live under ``app/logs/`` which is already gitignored. +(absolute binaries in the managed venv, no shell activation needed), the +working directory to spawn it in, and the URL to probe for readiness. This +module turns those specs into running processes: ``start`` spawns the server +(streaming its output to ``app/logs/<name>-server.log``, in the spec's cwd +when it has one — audio.cpp resolves model_specs/ relative to its process +working directory), records its pid, and polls the URL until it accepts +connections (model loads are slow, so the timeout is generous). With a spec +``identity`` the poll also verifies the server answers HTTP as that backend, +so readiness means "serving", not just "listening". + +Progress reporting goes through an optional ``progress`` callback (see +``_console_progress`` for the event shapes); the default callback prints the +same lines as before, so the plain-console flow is unchanged. ``stop`` +terminates the process group the hub started. + +Everything here can run in the plain console tail after the curses TUI +returns (matching the wizards' build/pip streaming) or behind the run view's +boot screen, which renders the same events. Pid/log files live under +``app/logs/`` which is already gitignored. """ import os @@ -19,9 +28,9 @@ import subprocess import sys import time from pathlib import Path -from typing import List +from typing import Callable, List, Optional -from backends import common +from backends import common, probe from backends.common import APP_DIR LOG_DIR = APP_DIR / "logs" @@ -34,6 +43,48 @@ SERVER_START_TIMEOUT = 600 # Grace period after SIGTERM before escalating to SIGKILL (POSIX). STOP_GRACE_SECONDS = 10 +# How often the start poll re-checks readiness (seconds). +POLL_INTERVAL = 1 + +# Progress callback: called with an event dict. KIND is one of: +# "starting" {name, argv, cwd, log_path, pid} spawned, waiting for boot +# "elapsed" {name, seconds} heartbeat while waiting +# "ready" {name, url} server is up and answering +# "exited" {name, returncode, log_tail} process exited while booting +# "timeout" {name, seconds, log_tail} readiness deadline elapsed +# "running" {name, url} already up (no spawn) +# "cancelled" {name} boot aborted via cancel +# "error" {message} could not spawn the executable +ProgressCallback = Optional[Callable[[dict], None]] + + +def _console_progress(event: dict) -> None: + """Print PROGRESS events as the plain-console output (the old behavior).""" + kind = event.get("kind") + if kind == "starting": + print(f"[INFO] starting {event['name']} server: {event['argv']}") + if event.get("cwd"): + print(f"[INFO] working directory: {event['cwd']}") + print(f"[INFO] pid {event['pid']}; logs: {event['log_path']}") + elif kind == "elapsed": + print(f"[INFO] still waiting for the server ({int(event['seconds'])}s)...") + elif kind == "ready": + print(f"[OK] {event['name']} server is up on {event['url']}") + elif kind == "running": + print(f"[INFO] {event['name']} server already running on {event['url']}") + elif kind == "cancelled": + print(f"[INFO] {event['name']} server start cancelled") + elif kind == "exited": + print(f"[ERROR] {event['name']} server exited with code " + f"{event['returncode']}") + _print_tail(event.get("log_tail")) + elif kind == "timeout": + print(f"[ERROR] {event['name']} server did not start within " + f"{int(event['seconds'])}s") + _print_tail(event.get("log_tail")) + elif kind == "error": + print(f"[ERROR] {event['message']}") + def _log_path(name: str) -> Path: return LOG_DIR / f"{name}-server.log" @@ -43,17 +94,20 @@ def _pid_path(name: str) -> Path: return LOG_DIR / f"{name}-server.pid" -def _tail_log(name: str, lines: int = 20) -> None: - """Print the last LINES of the server's log (best-effort).""" - path = _log_path(name) +def _read_log_tail(name: str, lines: int = 20) -> List[str]: + """Return the last LINES of the server's log (best-effort).""" try: - text = path.read_text(encoding="utf-8", errors="replace") + text = _log_path(name).read_text(encoding="utf-8", errors="replace") except OSError: - return - tail = "\n".join(text.splitlines()[-lines:]) + return [] + return text.splitlines()[-lines:] + + +def _print_tail(tail: List[str]) -> None: + """Print a log-tail event payload (used by the console callback).""" if tail: - print(f"--- last {lines} lines of {path} ---") - print(tail) + print(f"--- last {len(tail)} lines of the server log ---") + print("\n".join(tail)) print("---") @@ -123,24 +177,53 @@ def _kill_pid(pid: int) -> bool: return True -def start(spec) -> bool: +def _server_ready(spec) -> bool: + """True when the server described by SPEC is usable, not just listening. + + Without an IDENTITY this is the plain TCP-connect check. With one, the + server must also answer HTTP as that backend (``probe.identify_server``); + for the faster backend (whose model loads after the port opens) the + ``/health`` model_loaded flag must additionally be true. + """ + if not common.server_running(spec.url): + return False + identity = getattr(spec, "identity", None) + if identity is None: + return True + if probe.identify_server(spec.url) != identity: + return False + if identity == probe.IDENTITY_FASTER: + return probe.faster_model_loaded(spec.url) + return True + + +def start(spec, progress: ProgressCallback = None, + cancel=None) -> bool: """Start the server described by SPEC (a ``backends.ServerSpec``). - Spawns its argv with stdout/stderr to ``logs/<name>-server.log``, records - the pid, and polls ``common.server_running(spec.url)`` until it accepts - connections or ``SERVER_START_TIMEOUT`` elapses. Returns True when the - server is up; on timeout or early exit, prints the log tail and returns + Spawns its argv with stdout/stderr to ``logs/<name>-server.log``, in the + spec's CWD when it has one (audio.cpp discovers model_specs/ from its + process working directory), records the pid, and polls readiness — + ``_server_ready``, so an IDENTITY spec must actually answer HTTP — until + it is up or ``SERVER_START_TIMEOUT`` elapses. Returns True when the + server is up; on timeout or early exit reports the log tail and returns False. A no-op (True) when the server is already running. + + PROGRESS, when given, receives each boot event (see ProgressCallback); + the default ``_console_progress`` prints them, preserving the old + console output. CANCEL (a threading.Event) aborts the boot: the spawned + process is terminated and False is reported (event kind "cancelled"). """ + report = progress if progress is not None else _console_progress argv: List[str] = list(spec.argv) exe = Path(argv[0]) if not exe.exists(): - print(f"[ERROR] server executable not found: {exe}") - print(" run 'Set up a backend' for " - f"{spec.name!r} first.") + report({"kind": "error", + "message": f"server executable not found: {exe}. Run 'Set " + "up a backend' first."}) return False - if common.server_running(spec.url): - print(f"[INFO] {spec.name} server already running on {spec.url}") + if _server_ready(spec): + report({"kind": "running", "name": spec.name, "url": spec.url}) return True LOG_DIR.mkdir(parents=True, exist_ok=True) @@ -151,10 +234,11 @@ def start(spec) -> bool: except OSError: pass - print(f"[INFO] starting {spec.name} server: " - + " ".join(str(a) for a in argv)) + cwd = getattr(spec, "cwd", None) log_handle = _log_path(spec.name).open("w", encoding="utf-8") popen_kwargs = {"stdout": log_handle, "stderr": subprocess.STDOUT} + if cwd is not None: + popen_kwargs["cwd"] = str(cwd) if sys.platform == "win32": popen_kwargs["creationflags"] = \ subprocess.CREATE_NEW_PROCESS_GROUP # type: ignore[attr-defined] @@ -163,31 +247,51 @@ def start(spec) -> bool: try: proc = subprocess.Popen(argv, **popen_kwargs) except OSError as exc: - print(f"[ERROR] could not start server: {exc}") + report({"kind": "error", + "message": f"could not start server: {exc}"}) log_handle.close() return False pid_file.write_text(str(proc.pid), encoding="utf-8") - print(f"[INFO] pid {proc.pid}; logs: {_log_path(spec.name)}") - - deadline = time.time() + SERVER_START_TIMEOUT + report({"kind": "starting", "name": spec.name, + "argv": " ".join(str(a) for a in argv), + "cwd": str(cwd) if cwd is not None else None, + "log_path": str(_log_path(spec.name)), "pid": proc.pid}) + + started = time.time() + next_heartbeat = started + 15 + deadline = started + SERVER_START_TIMEOUT while time.time() < deadline: + if cancel is not None and cancel.is_set(): + # User cancelled while booting: kill what we spawned (the + # server we started is not left loading in the background). + _kill_pid(proc.pid) + try: + pid_file.unlink() + except OSError: + pass + report({"kind": "cancelled", "name": spec.name}) + return False if proc.poll() is not None: - print(f"[ERROR] {spec.name} server exited with code " - f"{proc.returncode}") - _tail_log(spec.name) + report({"kind": "exited", "name": spec.name, + "returncode": proc.returncode, + "log_tail": _read_log_tail(spec.name)}) try: pid_file.unlink() except OSError: pass return False - if common.server_running(spec.url): - print(f"[OK] {spec.name} server is up on {spec.url}") + if _server_ready(spec): + report({"kind": "ready", "name": spec.name, "url": spec.url}) return True - time.sleep(1) - print(f"[ERROR] {spec.name} server did not start within " - f"{SERVER_START_TIMEOUT}s") - _tail_log(spec.name) + if time.time() >= next_heartbeat: + report({"kind": "elapsed", + "seconds": time.time() - started}) + next_heartbeat += 15 + time.sleep(POLL_INTERVAL) + report({"kind": "timeout", "name": spec.name, + "seconds": SERVER_START_TIMEOUT, + "log_tail": _read_log_tail(spec.name)}) # Leave the pid file in place so stop() can kill it (it may still load). return False |
