Files
gpu-rent/tests/test_wait_backend_idle.py
T
Leonid Pershin f8bbde8ac0 Enhance backend recovery and diagnostics in installation and tuning scripts
- Improved the `install_swarm_comfy` function to handle empty backend states more effectively, introducing recovery mechanisms and enhanced logging for better visibility.
- Updated the `tune_swarm_perf` function to always sanitize backend FDS corruption, ensuring consistent performance tuning.
- Added new tests to validate the functionality of backend recovery and FDS sanitization, ensuring robustness in backend management.
2026-08-21 10:54:02 +03:00

160 lines
5.0 KiB
Python

"""Tests for wait_backend_idle READY semantics and install helpers."""
from gpu_rent.ready import _REMOTE_POLL
def test_remote_poll_empty_is_busy_not_ready():
assert 'bstat == "empty"' in _REMOTE_POLL
assert "BUSY backend=empty" in _REMOTE_POLL
def test_remote_poll_running_is_ready():
"""SwarmUI: running = healthy ready."""
assert 'bstat == "running"' in _REMOTE_POLL
assert "READY backend=running" in _REMOTE_POLL
def test_remote_poll_idle_suspended_is_ready():
"""Suspended idle backends are installed — ready for up (wake on gen)."""
assert "READY backend=idle" in _REMOTE_POLL
assert "BUSY backend=idle" not in _REMOTE_POLL
def test_remote_poll_loading_is_busy():
assert 'bstat in ("loading", "some_loading")' in _REMOTE_POLL
assert "Comfy стартует" in _REMOTE_POLL
def test_install_swarm_comfy_script_payload():
from importlib.resources import files
text = files("gpu_rent.remote").joinpath("install_swarm_comfy.py").read_text(
encoding="utf-8"
)
assert "InstallConfirmWS" in text
assert '"backend": "comfyui"' in text
assert '"models": "none"' in text
assert "modern_dark" in text
assert "backends present (idle/suspended)" in text
assert ".gpu-rent-comfy-installing" in text
assert "RestartBackends" in text
assert 'bstat == "errored"' in text
assert "sanitize_backends_fds" in text or "recover_empty_backends" in text
assert "AddNewBackend" in text
def test_wait_backend_idle_fail_fast_on_errored(monkeypatch):
from gpu_rent.errors import CloudError
from gpu_rent import ready
class Cfg:
pass
def fake_ssh(*a, **k):
return "BUSY backend=errored"
diag_calls = {"n": 0}
def fake_diag(*a, **k):
diag_calls["n"] += 1
return "DIAG ok"
monkeypatch.setattr(ready, "run_ssh", fake_ssh)
monkeypatch.setattr(ready, "collect_swarm_diagnostics", fake_diag)
try:
ready.wait_backend_idle(
Cfg(), "1.2.3.4", [].append, timeout=600.0, poll_every=0.01, errored_fail_sec=0.05
)
assert False, "expected CloudError"
except CloudError as exc:
assert "errored" in str(exc).lower()
assert "диагностик" in str(exc).lower() or "diag" in str(exc).lower()
assert diag_calls["n"] == 1
def test_swarm_diag_script_covers_api_and_journal():
from importlib.resources import files
text = files("gpu_rent.remote").joinpath("swarm_diag.py").read_text(encoding="utf-8")
assert "ListBackends" in text
assert "journalctl" in text
assert ".gpu-rent-last-diag.txt" in text
assert "nvidia-smi" in text
def test_verify_gpu_env_fail_fast_empty_dlbackend(monkeypatch):
import json
from gpu_rent.errors import CloudError
from gpu_rent.ready import verify_gpu_env
import gpu_rent.ssh_ops as ssh_ops
class Cfg:
enable_swarmui = True
payload = {
"ok": False,
"checks": [
{"name": "nvidia-smi", "required": True, "ok": True, "detail": "ok"},
{"name": "cuda", "required": True, "ok": True, "detail": "ok"},
{
"name": "torch",
"required": True,
"ok": False,
"detail": "ComfyUI venv python не найден (dlbackend пуст — SwarmUI Install не прогоняли)",
},
],
}
calls = {"n": 0}
def fake(*a, **k):
calls["n"] += 1
return json.dumps(payload)
monkeypatch.setattr(ssh_ops, "run_python", fake)
logs: list[str] = []
try:
verify_gpu_env(Cfg(), "1.2.3.4", logs.append, timeout=600.0, poll_every=0.1)
assert False, "expected CloudError"
except CloudError as exc:
assert "fail-fast" in str(exc).lower() or "torch" in str(exc).lower()
assert calls["n"] == 1
def test_verify_gpu_env_fail_fast_torch_no_cuda(monkeypatch):
import json
from gpu_rent.errors import CloudError
from gpu_rent.ready import verify_gpu_env
import gpu_rent.ssh_ops as ssh_ops
class Cfg:
enable_swarmui = True
payload = {
"ok": False,
"checks": [
{"name": "nvidia-smi", "required": True, "ok": True, "detail": "ok"},
{"name": "cuda", "required": True, "ok": True, "detail": "ok"},
{
"name": "torch",
"required": True,
"ok": False,
"detail": "torch=2.0 cuda=None available=false (CPU wheel / без cuda — не заживёт само)",
},
],
}
calls = {"n": 0}
def fake(*a, **k):
calls["n"] += 1
return json.dumps(payload)
monkeypatch.setattr(ssh_ops, "run_python", fake)
try:
verify_gpu_env(Cfg(), "1.2.3.4", [].append, timeout=600.0, poll_every=0.1)
assert False, "expected CloudError"
except CloudError:
pass
assert calls["n"] == 1