Files
gpu-rent/tests/test_perf_tiers.py
T
Leonid PershinandCursor ffe5a031b4 Warm Ollama into VRAM on up and tunnel so Assistent chat is not cold.
A 1-token /api/chat after tags (and again if /api/ps is empty) loads VL weights before the first message. Mid KEEP_ALIVE is 15m so a short image-gen burst does not unload the model.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-21 20:47:14 +03:00

88 lines
2.5 KiB
Python

from gpu_rent.perf_tiers import (
GpuInfo,
compute_cap_at_least,
ollama_env_lines,
ollama_tune_for,
swarm_tune_for,
tier_for_vram_mib,
)
def test_tier_boundaries():
assert tier_for_vram_mib(8 * 1024) == "low"
assert tier_for_vram_mib(16 * 1024) == "mid"
assert tier_for_vram_mib(23 * 1024) == "mid"
assert tier_for_vram_mib(24 * 1024) == "high"
assert tier_for_vram_mib(40 * 1024) == "high"
assert tier_for_vram_mib(48 * 1024) == "ultra"
assert tier_for_vram_mib(80 * 1024) == "ultra"
def test_compute_cap():
assert compute_cap_at_least("8.0", 8, 0)
assert compute_cap_at_least("8.9", 8, 0)
assert not compute_cap_at_least("7.5", 8, 0)
def test_ollama_high_reserves_vram_for_swarm():
info = GpuInfo(
name="NVIDIA A100-SXM4-40GB",
vram_mib=40960,
compute_cap="8.0",
uuid="GPU-1",
tier="high",
)
tune = ollama_tune_for(info)
assert tune.flash_attention
assert tune.num_parallel == 1
assert tune.max_loaded_models == 1
assert tune.kv_cache_type == "q8_0"
assert tune.gpu_overhead_bytes == 14 * 1024**3
env = "\n".join(ollama_env_lines(tune))
assert "OLLAMA_FLASH_ATTENTION=1" in env
assert "OLLAMA_GPU_OVERHEAD=" in env
assert "OLLAMA_CONTEXT_LENGTH=16384" in env
assert tune.context_length == 16384
def test_swarm_sage_on_ampere_mid():
info = GpuInfo(
name="NVIDIA GeForce RTX 4090",
vram_mib=24576,
compute_cap="8.9",
uuid="GPU-2",
tier="high",
)
st = swarm_tune_for(info)
assert st.use_sage_attention
assert "--use-sage-attention" in st.comfy_extra_args
assert st.install_triton_sage
def test_swarm_no_sage_on_low():
info = GpuInfo(
name="NVIDIA T4",
vram_mib=15360,
compute_cap="7.5",
uuid="GPU-3",
tier="low",
)
st = swarm_tune_for(info)
assert not st.use_sage_attention
assert st.comfy_extra_args == ""
def test_ollama_mid_4090_context_16k():
info = GpuInfo(
name="NVIDIA GeForce RTX 4090",
vram_mib=24564,
compute_cap="8.9",
uuid="GPU-4",
tier="mid",
)
tune = ollama_tune_for(info)
assert tune.context_length == 16384
assert tune.keep_alive == "15m"
assert "OLLAMA_CONTEXT_LENGTH=16384" in "\n".join(ollama_env_lines(tune))
assert "OLLAMA_KEEP_ALIVE=15m" in "\n".join(ollama_env_lines(tune))