"""Optional LLM runtimes (Ollama / llama.cpp) beside SwarmUI.""" from __future__ import annotations from dataclasses import dataclass from pathlib import Path from typing import Any from urllib.parse import unquote, urlparse import yaml from gpu_rent.paths import ( llamacpp_models_example_path, llamacpp_models_manifest_path, ollama_models_example_path, ollama_models_manifest_path, ) VALID_RUNTIMES = frozenset({"none", "ollama", "llamacpp"}) OLLAMA_PRESETS: dict[str, list[str]] = { "recommended": ["huihui_ai/qwen2.5-abliterate:7b"], "light": ["qwen2.5:3b"], "stock": ["qwen2.5:7b"], "alt": ["richardyoung/qwen2.5-7b-instruct-abliterated"], "empty": [], } OLLAMA_PRESET_LABELS: dict[str, str] = { "recommended": "Qwen2.5 7B abliterate (RU/EN, мало отказов, ~5GB)", "light": "qwen2.5:3b (быстрее, слабее)", "stock": "официальный qwen2.5:7b (больше цензуры)", "alt": "другой abliterate-пак 7B", "empty": "только runtime, без pull", "keep": "не менять ollama-models.yaml", } # Deprecated text blob — prefer menu helpers below. PRESET_HELP = "\n".join( f"{k} — {v}" for k, v in OLLAMA_PRESET_LABELS.items() if k != "keep" ) LLAMACPP_PRESETS: dict[str, list[str]] = { "recommended": [ "https://huggingface.co/bartowski/huihui-ai_Qwen2.5-7B-Instruct-abliterated-GGUF/resolve/main/huihui-ai_Qwen2.5-7B-Instruct-abliterated-Q4_K_M.gguf", ], "light": [ "https://huggingface.co/bartowski/Qwen2.5-3B-Instruct-GGUF/resolve/main/Qwen2.5-3B-Instruct-Q4_K_M.gguf", ], "stock": [ "https://huggingface.co/bartowski/Qwen2.5-7B-Instruct-GGUF/resolve/main/Qwen2.5-7B-Instruct-Q4_K_M.gguf", ], "empty": [], } LLAMACPP_PRESET_LABELS: dict[str, str] = { "recommended": "Qwen2.5 7B abliterate GGUF Q4_K_M (~4.7GB, мало отказов)", "light": "Qwen2.5 3B Instruct Q4_K_M (~2GB)", "stock": "официальный Qwen2.5 7B Instruct Q4_K_M", "empty": "только llama-server, GGUF вручную", "keep": "не менять llamacpp-models.yaml", } LLAMACPP_PRESET_HELP = "\n".join( f"{k} — {v}" for k, v in LLAMACPP_PRESET_LABELS.items() if k != "keep" ) LLM_RUNTIME_LABELS: dict[str, str] = { "none": "только SwarmUI", "ollama": "Ollama (+ pull моделей)", "llamacpp": "llama.cpp server (+ GGUF)", } WORKLOAD_LABELS: dict[str, str] = { "swarm": "только SwarmUI", "both": "SwarmUI + LLM", "llm": "только LLM (без SwarmUI)", } def llm_runtime_menu() -> list: from gpu_rent.prompts import MenuItem return [MenuItem(k, f"{k} — {LLM_RUNTIME_LABELS[k]}") for k in ("none", "ollama", "llamacpp")] def workload_menu() -> list: from gpu_rent.prompts import MenuItem return [MenuItem(k, WORKLOAD_LABELS[k]) for k in ("swarm", "both", "llm")] def ollama_preset_menu(*, include_keep: bool = False) -> list: from gpu_rent.prompts import MenuItem keys = list(OLLAMA_PRESETS.keys()) if include_keep: keys.append("keep") return [MenuItem(k, OLLAMA_PRESET_LABELS.get(k, k)) for k in keys] def llamacpp_preset_menu(*, include_keep: bool = False) -> list: from gpu_rent.prompts import MenuItem keys = list(LLAMACPP_PRESETS.keys()) if include_keep: keys.append("keep") return [MenuItem(k, LLAMACPP_PRESET_LABELS.get(k, k)) for k in keys] @dataclass(frozen=True) class OllamaModelEntry: name: str default: bool = False @dataclass(frozen=True) class LlamaCppModelEntry: url: str filename: str | None = None default: bool = False def normalize_runtime(value: str | None) -> str: raw = (value or "none").strip().lower().replace("-", "").replace("_", "") if raw in {"", "none", "off", "no", "0"}: return "none" if raw in {"ollama"}: return "ollama" if raw in {"llamacpp", "llama", "llamacppserver"}: return "llamacpp" raise ValueError(f"неизвестный LLM_RUNTIME={value!r}; жду none|ollama|llamacpp") def decide_runtime( *, flag: str | None, ollama_flag: bool, llamacpp_flag: bool, from_config: str, ) -> str: if ollama_flag and llamacpp_flag: raise ValueError("укажи только --ollama или --llamacpp, не оба") if ollama_flag: return "ollama" if llamacpp_flag: return "llamacpp" if flag is not None and str(flag).strip() != "": return normalize_runtime(flag) return normalize_runtime(from_config) def parse_ollama_models(path: Path) -> list[OllamaModelEntry]: if not path.is_file(): return [] raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {} if not isinstance(raw, dict): return [] items = raw.get("models") if items is None: return [] if not isinstance(items, list): raise ValueError(f"{path}: models должен быть списком") out: list[OllamaModelEntry] = [] for item in items: if isinstance(item, str): name = item.strip() if name: out.append(OllamaModelEntry(name=name)) continue if not isinstance(item, dict): continue name = str(item.get("name") or "").strip() if not name: continue out.append(OllamaModelEntry(name=name, default=bool(item.get("default")))) return out def gguf_filename_from_url(url: str) -> str: path = unquote(urlparse(url).path) name = path.rsplit("/", 1)[-1] if path else "" if name.lower().endswith(".gguf"): return name return "model.gguf" def parse_llamacpp_models(path: Path) -> list[LlamaCppModelEntry]: if not path.is_file(): return [] raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {} if not isinstance(raw, dict): return [] items = raw.get("models") if items is None: return [] if not isinstance(items, list): raise ValueError(f"{path}: models должен быть списком") out: list[LlamaCppModelEntry] = [] for item in items: if isinstance(item, str): url = item.strip() if url: out.append(LlamaCppModelEntry(url=url)) continue if not isinstance(item, dict): continue url = str(item.get("url") or "").strip() if not url: continue fname = item.get("filename") filename = str(fname).strip() if fname else None out.append( LlamaCppModelEntry( url=url, filename=filename or None, default=bool(item.get("default")), ) ) return out def write_ollama_models_preset(path: Path, preset: str) -> None: key = (preset or "recommended").strip().lower() if key not in OLLAMA_PRESETS: raise ValueError(f"пресет {preset!r}; варианты: {', '.join(OLLAMA_PRESETS)}") names = OLLAMA_PRESETS[key] lines = [ "# Локальный манифест Ollama (не коммить). Пример: ollama-models.example.yaml", "# name = точный тег для `ollama pull`. Пустой models: [] — без pull.", "models:", ] if not names: lines.append(" []") else: for i, name in enumerate(names): lines.append(f" - name: {name}") if i == 0: lines.append(" default: true") path.write_text("\n".join(lines) + "\n", encoding="utf-8") def write_llamacpp_models_preset(path: Path, preset: str) -> None: key = (preset or "recommended").strip().lower() if key not in LLAMACPP_PRESETS: raise ValueError(f"пресет {preset!r}; варианты: {', '.join(LLAMACPP_PRESETS)}") urls = LLAMACPP_PRESETS[key] lines = [ "# Локальный манифест llama.cpp GGUF (не коммить). Пример: llamacpp-models.example.yaml", "# url = прямой HTTPS на .gguf. Пустой models: [] — без скачивания.", "models:", ] if not urls: lines.append(" []") else: for i, url in enumerate(urls): lines.append(f" - url: {url}") if i == 0: lines.append(" default: true") path.write_text("\n".join(lines) + "\n", encoding="utf-8") def ensure_ollama_manifest_from_example() -> Path: dest = ollama_models_manifest_path() if dest.is_file(): return dest example = ollama_models_example_path() if example.is_file(): dest.write_text(example.read_text(encoding="utf-8"), encoding="utf-8") else: write_ollama_models_preset(dest, "recommended") return dest def ensure_llamacpp_manifest_from_example() -> Path: dest = llamacpp_models_manifest_path() if dest.is_file(): return dest example = llamacpp_models_example_path() if example.is_file(): dest.write_text(example.read_text(encoding="utf-8"), encoding="utf-8") else: write_llamacpp_models_preset(dest, "recommended") return dest def llm_local_port(cfg: Any) -> int | None: runtime = normalize_runtime(getattr(cfg, "llm_runtime", "none")) if runtime == "ollama": return int(getattr(cfg, "ollama_local_port", 17811)) if runtime == "llamacpp": return int(getattr(cfg, "llamacpp_local_port", 17812)) return None def llm_remote_port(runtime: str) -> int | None: runtime = normalize_runtime(runtime) if runtime == "ollama": return 11434 if runtime == "llamacpp": return 8080 return None def append_vars_llm_runtime(vars_file: Path, runtime: str) -> None: from gpu_rent.varsfile import upsert_vars upsert_vars(vars_file, {"LLM_RUNTIME": normalize_runtime(runtime)})