- Removed references to llamacpp from configuration files, scripts, and documentation, streamlining the LLM setup process to focus solely on Ollama. - Updated environment variables and paths to eliminate llamacpp-related entries, ensuring clarity in the configuration. - Adjusted CLI commands and help messages to reflect the removal of llamacpp, enhancing user experience and reducing confusion. - Revised documentation to provide clear guidance on using Ollama exclusively, including updates to setup instructions and runtime options.
290 lines
10 KiB
Python
290 lines
10 KiB
Python
"""Remote idle-killer: stdlib only. Runs on the VM via systemd timer."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import ssl
|
|
import time
|
|
import urllib.error
|
|
import urllib.request
|
|
from pathlib import Path
|
|
|
|
DATA = Path("/mnt/swarm_data")
|
|
CREDS = Path("/root/.gpu-rent/idle-killer.json")
|
|
HOLD = DATA / ".gpu-rent-hold-until"
|
|
IDLE_SINCE = DATA / ".gpu-rent-idle-since"
|
|
SWARM_DOWN_SINCE = DATA / ".gpu-rent-swarm-down-since"
|
|
ARMED = DATA / ".gpu-rent-killer-armed"
|
|
LOG = DATA / ".gpu-rent-killer.log"
|
|
SERVER_ID_FILE = DATA / ".gpu-rent-server-id"
|
|
# After this many seconds of continuous Swarm unreachable → treat as idle (don't bill forever).
|
|
SWARM_UNREACHABLE_IDLE_SEC = 2 * 60 * 60
|
|
|
|
|
|
def log(msg: str) -> None:
|
|
line = f"{time.strftime('%Y-%m-%dT%H:%M:%SZ', time.gmtime())} {msg}\n"
|
|
try:
|
|
LOG.parent.mkdir(parents=True, exist_ok=True)
|
|
with LOG.open("a", encoding="utf-8") as fh:
|
|
fh.write(line)
|
|
except OSError:
|
|
pass
|
|
print(line.rstrip())
|
|
|
|
|
|
def read_ts(path: Path) -> float | None:
|
|
if not path.is_file():
|
|
return None
|
|
try:
|
|
return float(path.read_text(encoding="utf-8").strip().split()[0])
|
|
except (OSError, ValueError, IndexError):
|
|
return None
|
|
|
|
|
|
def write_ts(path: Path, value: float | None) -> None:
|
|
if value is None:
|
|
try:
|
|
path.unlink(missing_ok=True)
|
|
except OSError:
|
|
pass
|
|
return
|
|
path.write_text(f"{int(value)}\n", encoding="utf-8")
|
|
|
|
|
|
def swarm_busy(swarm_url: str, timeout: float = 8.0) -> tuple[bool, str]:
|
|
"""Return (busy, detail).
|
|
|
|
Unreachable UI is busy during boot, but after SWARM_UNREACHABLE_IDLE_SEC of
|
|
continuous failure we stop counting it as busy so idle clock can run / delete.
|
|
"""
|
|
ctx = ssl.create_default_context()
|
|
try:
|
|
req = urllib.request.Request(
|
|
f"{swarm_url.rstrip('/')}/API/GetNewSession",
|
|
data=b"{}",
|
|
headers={"Content-Type": "application/json"},
|
|
method="POST",
|
|
)
|
|
with urllib.request.urlopen(req, timeout=timeout, context=ctx) as resp:
|
|
session = json.loads(resp.read().decode("utf-8"))
|
|
sid = session.get("session_id")
|
|
if not sid:
|
|
return True, "no session_id"
|
|
body = json.dumps({"session_id": sid}).encode("utf-8")
|
|
req2 = urllib.request.Request(
|
|
f"{swarm_url.rstrip('/')}/API/GetCurrentStatus",
|
|
data=body,
|
|
headers={"Content-Type": "application/json"},
|
|
method="POST",
|
|
)
|
|
with urllib.request.urlopen(req2, timeout=timeout, context=ctx) as resp:
|
|
data = json.loads(resp.read().decode("utf-8"))
|
|
except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError, json.JSONDecodeError, OSError) as exc:
|
|
now = time.time()
|
|
since = read_ts(SWARM_DOWN_SINCE)
|
|
if since is None:
|
|
write_ts(SWARM_DOWN_SINCE, now)
|
|
return True, f"swarm unreachable (clock start): {exc}"
|
|
down_for = now - since
|
|
if down_for >= SWARM_UNREACHABLE_IDLE_SEC:
|
|
return False, f"swarm unreachable {int(down_for)}s >= {SWARM_UNREACHABLE_IDLE_SEC}s — allow idle"
|
|
return True, f"swarm unreachable {int(down_for)}s / {SWARM_UNREACHABLE_IDLE_SEC}s: {exc}"
|
|
|
|
# Reachable again — clear down clock.
|
|
write_ts(SWARM_DOWN_SINCE, None)
|
|
|
|
status = data.get("status") or {}
|
|
backend = data.get("backend_status") or {}
|
|
waiting = int(status.get("waiting_gens") or 0)
|
|
live = int(status.get("live_gens") or 0)
|
|
loading = int(status.get("loading_models") or 0)
|
|
bstat = str(backend.get("status") or "unknown").lower()
|
|
if waiting or live or loading:
|
|
return True, f"queue waiting={waiting} live={live} loading={loading}"
|
|
if bstat not in {"idle", "disabled", "all_disabled", "empty"}:
|
|
return True, f"backend={bstat}"
|
|
return False, f"idle backend={bstat}"
|
|
|
|
|
|
def llm_busy(timeout: float = 3.0) -> tuple[bool, str]:
|
|
"""Ollama pull / loaded models count as busy."""
|
|
pull_marker = DATA / ".gpu-rent-ollama-pulling"
|
|
if pull_marker.is_file():
|
|
try:
|
|
ts = float(pull_marker.read_text(encoding="utf-8").strip().split()[0])
|
|
age = time.time() - ts
|
|
except (OSError, ValueError, IndexError):
|
|
age = 0.0
|
|
ts = 0.0
|
|
# Stale marker after SSH kill / crash — don't block billing forever.
|
|
max_age = 45 * 60
|
|
if age > max_age:
|
|
try:
|
|
pull_marker.unlink(missing_ok=True)
|
|
except OSError:
|
|
pass
|
|
log(f"cleared stale ollama-pulling marker age={int(age)}s")
|
|
else:
|
|
return True, f"ollama pulling ({int(age)}s)"
|
|
ctx = ssl.create_default_context()
|
|
# Ollama: any running model
|
|
try:
|
|
req = urllib.request.Request("http://127.0.0.1:11434/api/ps", method="GET")
|
|
with urllib.request.urlopen(req, timeout=timeout, context=ctx) as resp:
|
|
data = json.loads(resp.read().decode("utf-8"))
|
|
models = data.get("models") or []
|
|
if models:
|
|
names = ",".join(str(m.get("name") or "?") for m in models[:3])
|
|
return True, f"ollama running {names}"
|
|
except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError, json.JSONDecodeError, OSError):
|
|
pass
|
|
return False, "llm idle"
|
|
|
|
|
|
def keystone_token(creds: dict) -> tuple[str, str]:
|
|
"""Return (token, compute_url)."""
|
|
auth = {
|
|
"auth": {
|
|
"identity": {
|
|
"methods": ["application_credential"],
|
|
"application_credential": {
|
|
"id": creds["application_credential_id"],
|
|
"secret": creds["application_credential_secret"],
|
|
},
|
|
}
|
|
}
|
|
}
|
|
url = creds["auth_url"].rstrip("/") + "/auth/tokens"
|
|
body = json.dumps(auth).encode("utf-8")
|
|
req = urllib.request.Request(
|
|
url,
|
|
data=body,
|
|
headers={"Content-Type": "application/json"},
|
|
method="POST",
|
|
)
|
|
with urllib.request.urlopen(req, timeout=30) as resp:
|
|
token = resp.headers.get("X-Subject-Token")
|
|
payload = json.loads(resp.read().decode("utf-8"))
|
|
if not token:
|
|
raise RuntimeError("нет X-Subject-Token")
|
|
catalog = ((payload.get("token") or {}).get("catalog")) or []
|
|
region = (creds.get("region_name") or "").lower()
|
|
compute = ""
|
|
for svc in catalog:
|
|
if svc.get("type") != "compute":
|
|
continue
|
|
for ep in svc.get("endpoints") or []:
|
|
if ep.get("interface") != "public":
|
|
continue
|
|
if region and str(ep.get("region") or "").lower() != region:
|
|
continue
|
|
compute = str(ep.get("url") or "").rstrip("/")
|
|
break
|
|
if compute:
|
|
break
|
|
if not compute:
|
|
raise RuntimeError("compute endpoint не найден в catalog")
|
|
# Prefer v2.1
|
|
if "/v2/" in compute and "/v2.1" not in compute:
|
|
compute = compute.replace("/v2/", "/v2.1/")
|
|
return token, compute
|
|
|
|
|
|
def delete_server(creds: dict, server_id: str) -> None:
|
|
token, compute = keystone_token(creds)
|
|
url = f"{compute}/servers/{server_id}"
|
|
req = urllib.request.Request(
|
|
url,
|
|
headers={"X-Auth-Token": token},
|
|
method="DELETE",
|
|
)
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=60) as resp:
|
|
code = getattr(resp, "status", 204)
|
|
log(f"DELETE {server_id} -> {code}")
|
|
except urllib.error.HTTPError as exc:
|
|
if exc.code in (404, 410):
|
|
log(f"server already gone ({exc.code})")
|
|
return
|
|
raise
|
|
|
|
|
|
def main() -> int:
|
|
if not ARMED.is_file():
|
|
log("not armed")
|
|
return 0
|
|
if not CREDS.is_file():
|
|
log("blind: no creds")
|
|
return 0
|
|
try:
|
|
creds = json.loads(CREDS.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError) as exc:
|
|
log(f"blind: bad creds ({exc})")
|
|
return 0
|
|
|
|
now = time.time()
|
|
grace_from = float(creds.get("grace_from") or 0)
|
|
grace_minutes = float(creds.get("grace_minutes") or 45)
|
|
if now < grace_from + grace_minutes * 60:
|
|
left = int(grace_from + grace_minutes * 60 - now)
|
|
log(f"grace {left}s left")
|
|
return 0
|
|
|
|
hold_until = read_ts(HOLD)
|
|
if hold_until and now < hold_until:
|
|
log(f"hold until {int(hold_until)}")
|
|
return 0
|
|
|
|
swarm_url = str(creds.get("swarm_url") or "http://127.0.0.1:7801")
|
|
if (DATA / ".gpu-rent-llm-only").is_file():
|
|
busy, detail = False, "llm-only (swarm skip)"
|
|
else:
|
|
busy, detail = swarm_busy(swarm_url)
|
|
if busy:
|
|
write_ts(IDLE_SINCE, None)
|
|
log(f"busy: {detail}")
|
|
return 0
|
|
|
|
llm_is_busy, llm_detail = llm_busy()
|
|
if llm_is_busy:
|
|
write_ts(IDLE_SINCE, None)
|
|
log(f"busy: {llm_detail}")
|
|
return 0
|
|
|
|
idle_minutes = float(creds.get("idle_minutes") or 30)
|
|
since = read_ts(IDLE_SINCE)
|
|
if since is None:
|
|
write_ts(IDLE_SINCE, now)
|
|
log(f"idle clock start ({detail})")
|
|
return 0
|
|
|
|
elapsed = now - since
|
|
need = idle_minutes * 60
|
|
if elapsed < need:
|
|
log(f"idle {int(elapsed)}s / {int(need)}s ({detail})")
|
|
return 0
|
|
|
|
server_id = str(creds.get("server_id") or "").strip()
|
|
if not server_id and SERVER_ID_FILE.is_file():
|
|
server_id = SERVER_ID_FILE.read_text(encoding="utf-8").strip()
|
|
if not server_id:
|
|
log("no server_id")
|
|
return 1
|
|
|
|
log(f"idle {int(elapsed)}s >= {int(need)}s — delete {server_id}")
|
|
try:
|
|
delete_server(creds, server_id)
|
|
except Exception as exc:
|
|
log(f"delete failed: {exc}")
|
|
return 1
|
|
write_ts(IDLE_SINCE, None)
|
|
try:
|
|
ARMED.unlink(missing_ok=True)
|
|
except OSError:
|
|
pass
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|