Implement idle-killer enhancements and swarm management improvements

- Marked critical bugs as resolved in the review documentation, including changes to the `arm_idle_killer` function to raise errors on credential creation failures and ensure proper file permissions for JSON credentials.
- Introduced a new `_try_arm_idle_killer` function in `provision.py` to manage idle-killer state more effectively, ensuring it arms correctly during provisioning.
- Updated the `swarm_busy` function in `remote/idle_killer.py` to allow idle state after a specified duration of Swarm unavailability, preventing unnecessary billing.
- Enhanced performance tuning logic in `tune_swarm_perf.py` to ensure proper handling of pip installation success before applying extra arguments.
- Added tests to validate the new idle-killer behavior and swarm management logic, ensuring robustness in handling idle states and error conditions.
This commit is contained in:
Leonid Pershin
2026-08-21 07:04:09 +03:00
parent d409e2e154
commit 1ec615c03e
7 changed files with 268 additions and 104 deletions
+88
View File
@@ -0,0 +1,88 @@
"""Tests for remote tune_swarm_perf pip_ok / ExtraArgs gating."""
from __future__ import annotations
import importlib.util
import json
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
REMOTE = ROOT / "src" / "gpu_rent" / "remote" / "tune_swarm_perf.py"
def _load():
spec = importlib.util.spec_from_file_location("tune_swarm_perf", REMOTE)
assert spec and spec.loader
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
return mod
def test_pip_fail_skips_extra_args(tmp_path, monkeypatch):
mod = _load()
data = tmp_path
backends = data / "Data" / "Backends.fds"
backends.parent.mkdir(parents=True)
backends.write_text("ExtraArgs: \n", encoding="utf-8")
gpu_json = data / ".gpu-rent-gpu.json"
gpu_json.write_text(
json.dumps(
{
"vram_mib": 24576,
"compute_cap": "8.9",
"uuid": "gpu-1",
"name": "RTX",
}
),
encoding="utf-8",
)
pip = data / "fake-pip"
pip.write_text("#!/bin/sh\n", encoding="utf-8")
monkeypatch.setattr(mod, "DATA", data)
monkeypatch.setattr(mod, "GPU_JSON", gpu_json)
monkeypatch.setattr(mod, "MARKER", data / ".gpu-rent-perf-tuned")
monkeypatch.setattr(mod, "BACKENDS", backends)
monkeypatch.setattr(mod, "find_pip", lambda: pip)
monkeypatch.setattr(mod, "pip_install_sage", lambda _p: False)
assert mod.main() == 0
marker = json.loads((data / ".gpu-rent-perf-tuned").read_text(encoding="utf-8"))
assert marker["pip_ok"] is False
assert marker["extra_args"] == ""
assert "--use-sage-attention" not in backends.read_text(encoding="utf-8")
def test_pip_ok_patches_extra_args(tmp_path, monkeypatch):
mod = _load()
data = tmp_path
backends = data / "Data" / "Backends.fds"
backends.parent.mkdir(parents=True)
backends.write_text("ExtraArgs: \n", encoding="utf-8")
gpu_json = data / ".gpu-rent-gpu.json"
gpu_json.write_text(
json.dumps(
{
"vram_mib": 24576,
"compute_cap": "8.9",
"uuid": "gpu-1",
"name": "RTX",
}
),
encoding="utf-8",
)
pip = data / "fake-pip"
pip.write_text("#!/bin/sh\n", encoding="utf-8")
monkeypatch.setattr(mod, "DATA", data)
monkeypatch.setattr(mod, "GPU_JSON", gpu_json)
monkeypatch.setattr(mod, "MARKER", data / ".gpu-rent-perf-tuned")
monkeypatch.setattr(mod, "BACKENDS", backends)
monkeypatch.setattr(mod, "find_pip", lambda: pip)
monkeypatch.setattr(mod, "pip_install_sage", lambda _p: True)
assert mod.main() == 0
marker = json.loads((data / ".gpu-rent-perf-tuned").read_text(encoding="utf-8"))
assert marker["pip_ok"] is True
assert "--use-sage-attention" in marker["extra_args"]
assert "--use-sage-attention" in backends.read_text(encoding="utf-8")