Refactor backend status handling and improve idle management

- Updated the idle-killer logic to treat SwarmUI `empty` and `disabled` states as busy, preventing unnecessary idle time during provisioning.
- Enhanced the `wait_backend_idle` function to recognize suspended backends as ready, improving resource utilization and user feedback.
- Refined the `install_swarm_comfy` script to skip installation when backends are already present, streamlining the setup process.
- Improved the `resolve_llm_runtime` function to prioritize live configuration over stale state notes, ensuring accurate runtime detection.
- Added tests to validate the new backend status handling and idle management logic, ensuring robustness and reliability.
This commit is contained in:
Leonid Pershin
2026-08-21 10:07:16 +03:00
parent 26f3be6e96
commit 1785ab369c
13 changed files with 302 additions and 90 deletions
+47 -6
View File
@@ -9,11 +9,15 @@ def test_remote_poll_empty_is_busy_not_ready():
def test_remote_poll_running_is_ready():
"""SwarmUI: running = healthy ready; idle = suspended (cannot generate)."""
"""SwarmUI: running = healthy ready."""
assert 'bstat == "running"' in _REMOTE_POLL
assert "READY backend=running" in _REMOTE_POLL
assert "READY backend=idle" not in _REMOTE_POLL
assert "BUSY backend=idle" in _REMOTE_POLL
def test_remote_poll_idle_suspended_is_ready():
"""Suspended idle backends are installed — ready for up (wake on gen)."""
assert "READY backend=idle" in _REMOTE_POLL
assert "BUSY backend=idle" not in _REMOTE_POLL
def test_remote_poll_loading_is_busy():
@@ -31,9 +35,8 @@ def test_install_swarm_comfy_script_payload():
assert '"backend": "comfyui"' in text
assert '"models": "none"' in text
assert "modern_dark" in text
assert "detect_stage" in text
assert "dlbackend=" in text
assert 'end="\\r"' in text or 'end="\\r"' in text
assert "backends present (idle/suspended)" in text
assert ".gpu-rent-comfy-installing" in text
def test_verify_gpu_env_fail_fast_empty_dlbackend(monkeypatch):
@@ -73,3 +76,41 @@ def test_verify_gpu_env_fail_fast_empty_dlbackend(monkeypatch):
except CloudError as exc:
assert "fail-fast" in str(exc).lower() or "torch" in str(exc).lower()
assert calls["n"] == 1
def test_verify_gpu_env_fail_fast_torch_no_cuda(monkeypatch):
import json
from gpu_rent.errors import CloudError
from gpu_rent.ready import verify_gpu_env
import gpu_rent.ssh_ops as ssh_ops
class Cfg:
enable_swarmui = True
payload = {
"ok": False,
"checks": [
{"name": "nvidia-smi", "required": True, "ok": True, "detail": "ok"},
{"name": "cuda", "required": True, "ok": True, "detail": "ok"},
{
"name": "torch",
"required": True,
"ok": False,
"detail": "torch=2.0 cuda=None available=false (CPU wheel / без cuda — не заживёт само)",
},
],
}
calls = {"n": 0}
def fake(*a, **k):
calls["n"] += 1
return json.dumps(payload)
monkeypatch.setattr(ssh_ops, "run_python", fake)
try:
verify_gpu_env(Cfg(), "1.2.3.4", [].append, timeout=600.0, poll_every=0.1)
assert False, "expected CloudError"
except CloudError:
pass
assert calls["n"] == 1