- Improved the `install_swarm_comfy` function to handle empty backend states more effectively, introducing recovery mechanisms and enhanced logging for better visibility. - Updated the `tune_swarm_perf` function to always sanitize backend FDS corruption, ensuring consistent performance tuning. - Added new tests to validate the functionality of backend recovery and FDS sanitization, ensuring robustness in backend management.
160 lines
5.0 KiB
Python
160 lines
5.0 KiB
Python
"""Tests for wait_backend_idle READY semantics and install helpers."""
|
|
|
|
from gpu_rent.ready import _REMOTE_POLL
|
|
|
|
|
|
def test_remote_poll_empty_is_busy_not_ready():
|
|
assert 'bstat == "empty"' in _REMOTE_POLL
|
|
assert "BUSY backend=empty" in _REMOTE_POLL
|
|
|
|
|
|
def test_remote_poll_running_is_ready():
|
|
"""SwarmUI: running = healthy ready."""
|
|
assert 'bstat == "running"' in _REMOTE_POLL
|
|
assert "READY backend=running" in _REMOTE_POLL
|
|
|
|
|
|
def test_remote_poll_idle_suspended_is_ready():
|
|
"""Suspended idle backends are installed — ready for up (wake on gen)."""
|
|
assert "READY backend=idle" in _REMOTE_POLL
|
|
assert "BUSY backend=idle" not in _REMOTE_POLL
|
|
|
|
|
|
def test_remote_poll_loading_is_busy():
|
|
assert 'bstat in ("loading", "some_loading")' in _REMOTE_POLL
|
|
assert "Comfy стартует" in _REMOTE_POLL
|
|
|
|
|
|
def test_install_swarm_comfy_script_payload():
|
|
from importlib.resources import files
|
|
|
|
text = files("gpu_rent.remote").joinpath("install_swarm_comfy.py").read_text(
|
|
encoding="utf-8"
|
|
)
|
|
assert "InstallConfirmWS" in text
|
|
assert '"backend": "comfyui"' in text
|
|
assert '"models": "none"' in text
|
|
assert "modern_dark" in text
|
|
assert "backends present (idle/suspended)" in text
|
|
assert ".gpu-rent-comfy-installing" in text
|
|
assert "RestartBackends" in text
|
|
assert 'bstat == "errored"' in text
|
|
assert "sanitize_backends_fds" in text or "recover_empty_backends" in text
|
|
assert "AddNewBackend" in text
|
|
|
|
|
|
def test_wait_backend_idle_fail_fast_on_errored(monkeypatch):
|
|
from gpu_rent.errors import CloudError
|
|
from gpu_rent import ready
|
|
|
|
class Cfg:
|
|
pass
|
|
|
|
def fake_ssh(*a, **k):
|
|
return "BUSY backend=errored"
|
|
|
|
diag_calls = {"n": 0}
|
|
|
|
def fake_diag(*a, **k):
|
|
diag_calls["n"] += 1
|
|
return "DIAG ok"
|
|
|
|
monkeypatch.setattr(ready, "run_ssh", fake_ssh)
|
|
monkeypatch.setattr(ready, "collect_swarm_diagnostics", fake_diag)
|
|
try:
|
|
ready.wait_backend_idle(
|
|
Cfg(), "1.2.3.4", [].append, timeout=600.0, poll_every=0.01, errored_fail_sec=0.05
|
|
)
|
|
assert False, "expected CloudError"
|
|
except CloudError as exc:
|
|
assert "errored" in str(exc).lower()
|
|
assert "диагностик" in str(exc).lower() or "diag" in str(exc).lower()
|
|
assert diag_calls["n"] == 1
|
|
|
|
|
|
def test_swarm_diag_script_covers_api_and_journal():
|
|
from importlib.resources import files
|
|
|
|
text = files("gpu_rent.remote").joinpath("swarm_diag.py").read_text(encoding="utf-8")
|
|
assert "ListBackends" in text
|
|
assert "journalctl" in text
|
|
assert ".gpu-rent-last-diag.txt" in text
|
|
assert "nvidia-smi" in text
|
|
|
|
|
|
def test_verify_gpu_env_fail_fast_empty_dlbackend(monkeypatch):
|
|
import json
|
|
|
|
from gpu_rent.errors import CloudError
|
|
from gpu_rent.ready import verify_gpu_env
|
|
import gpu_rent.ssh_ops as ssh_ops
|
|
|
|
class Cfg:
|
|
enable_swarmui = True
|
|
|
|
payload = {
|
|
"ok": False,
|
|
"checks": [
|
|
{"name": "nvidia-smi", "required": True, "ok": True, "detail": "ok"},
|
|
{"name": "cuda", "required": True, "ok": True, "detail": "ok"},
|
|
{
|
|
"name": "torch",
|
|
"required": True,
|
|
"ok": False,
|
|
"detail": "ComfyUI venv python не найден (dlbackend пуст — SwarmUI Install не прогоняли)",
|
|
},
|
|
],
|
|
}
|
|
calls = {"n": 0}
|
|
|
|
def fake(*a, **k):
|
|
calls["n"] += 1
|
|
return json.dumps(payload)
|
|
|
|
monkeypatch.setattr(ssh_ops, "run_python", fake)
|
|
logs: list[str] = []
|
|
try:
|
|
verify_gpu_env(Cfg(), "1.2.3.4", logs.append, timeout=600.0, poll_every=0.1)
|
|
assert False, "expected CloudError"
|
|
except CloudError as exc:
|
|
assert "fail-fast" in str(exc).lower() or "torch" in str(exc).lower()
|
|
assert calls["n"] == 1
|
|
|
|
|
|
def test_verify_gpu_env_fail_fast_torch_no_cuda(monkeypatch):
|
|
import json
|
|
|
|
from gpu_rent.errors import CloudError
|
|
from gpu_rent.ready import verify_gpu_env
|
|
import gpu_rent.ssh_ops as ssh_ops
|
|
|
|
class Cfg:
|
|
enable_swarmui = True
|
|
|
|
payload = {
|
|
"ok": False,
|
|
"checks": [
|
|
{"name": "nvidia-smi", "required": True, "ok": True, "detail": "ok"},
|
|
{"name": "cuda", "required": True, "ok": True, "detail": "ok"},
|
|
{
|
|
"name": "torch",
|
|
"required": True,
|
|
"ok": False,
|
|
"detail": "torch=2.0 cuda=None available=false (CPU wheel / без cuda — не заживёт само)",
|
|
},
|
|
],
|
|
}
|
|
calls = {"n": 0}
|
|
|
|
def fake(*a, **k):
|
|
calls["n"] += 1
|
|
return json.dumps(payload)
|
|
|
|
monkeypatch.setattr(ssh_ops, "run_python", fake)
|
|
try:
|
|
verify_gpu_env(Cfg(), "1.2.3.4", [].append, timeout=600.0, poll_every=0.1)
|
|
assert False, "expected CloudError"
|
|
except CloudError:
|
|
pass
|
|
assert calls["n"] == 1
|