diff --git a/.gitignore b/.gitignore index addedbe..1661645 100644 --- a/.gitignore +++ b/.gitignore @@ -4,7 +4,6 @@ !.env.example gpu-rent.vars ollama-models.yaml -llamacpp-models.yaml models.yaml extensions.yaml diff --git a/README.md b/README.md index 6ba66fe..bcb2502 100644 --- a/README.md +++ b/README.md @@ -55,8 +55,8 @@ Unix: `./gpu-rent.sh …` (один раз `chmod +x gpu-rent.sh`). | `doctor` | Preflight без create | | `flavors` / `dry-run` | Что выберет / план без денег | | `up` / `up --yes` | GPU + seed + туннель; без `--yes` — меню Selectel/LLM | -| `up --ollama` / `--llamacpp` | + LLM (см. [docs/llm.md](docs/llm.md)) | -| `up --llm-only --llamacpp` | только LLM, без SwarmUI | +| `up --ollama` | + Ollama (см. [docs/llm.md](docs/llm.md)) | +| `up --llm-only --ollama` | только Ollama, без SwarmUI | | `tunnel` / `open` | Снова UI / браузер | | `hold` / `status` | Пауза killer / состояние | | `stop` / `destroy --i-understand-data-loss` | Стоп GPU / + диски | @@ -73,7 +73,7 @@ Unix: `./gpu-rent.sh …` (один раз `chmod +x gpu-rent.sh`). | `.env` | из `env.example` — `OS_*`, Civitai (не в git) | | `gpu-rent.vars` | из example — несекретные дефолты, лаунчер | | `models.yaml` / `extensions.yaml` | манифесты (из `*.example.yaml`) | -| `ollama-models.yaml` / `llamacpp-models.yaml` | LLM (из example; gitignore) | +| `ollama-models.yaml` | Ollama-модели (из example; gitignore) | | `Models/` … | локальный push на `up` | | `.gpu-rent/` | state, SSH-ключ (gitignore) | @@ -87,7 +87,7 @@ Unix: `./gpu-rent.sh …` (один раз `chmod +x gpu-rent.sh`). | --- | --- | | [docs/setup.md](docs/setup.md) | **Пошаговая подготовка** до первого `up` | | [docs/cli.md](docs/cli.md) | Все команды и переменные | -| [docs/llm.md](docs/llm.md) | Ollama / llama.cpp | +| [docs/llm.md](docs/llm.md) | Ollama (opt-in LLM) | | [docs/models.md](docs/models.md) | Civitai + папка `Models/` | | [docs/spike-notes.md](docs/spike-notes.md) | Чеклист первого живого прогона | | [docs/README.md](docs/README.md) | Оглавление всего `docs/` | diff --git a/docs/README.md b/docs/README.md index 8558d11..cca8c47 100644 --- a/docs/README.md +++ b/docs/README.md @@ -20,7 +20,7 @@ CLI поднимает прерываемый GPU в Selectel, держит Swar | Git-расширения SwarmUI/Comfy | [extensions.md](extensions.md) | | Word-list промптов | [autocomplete.md](autocomplete.md) | | Push/pull папок | [local-folders.md](local-folders.md) | -| Ollama / llama.cpp | [llm.md](llm.md) | +| Ollama | [llm.md](llm.md) | | Capture ссылок с VM | [cli.md](cli.md) (`capture`) + [models.md](models.md) | | Как устроены диски и killer | [architecture.md](architecture.md) | | Контракт Selectel | [selectel.md](selectel.md) | diff --git a/docs/architecture.md b/docs/architecture.md index 48201f8..11550a7 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -19,7 +19,7 @@ │ SwarmUI : 127.0.0.1:7801 │ │ idle-killer: очередь / hold / LLM busy → delete this server │ │ app cred: DELETE/GET только этот server_id (fail closed) │ -│ optional: Ollama :11434 / llama.cpp :8080 (туннель 17811/12)│ +│ optional: Ollama :11434 (туннель 17811) │ │ │ │ boot volume (network) ОС + NVIDIA + SwarmUI нативно + snapshot │ │ data volume (network) Models, Output, Data, workflows │ @@ -49,7 +49,7 @@ | `notify` | Toast/звук при backend Idle | | `tunnel` | sshtunnel + Nova EXPIRED watchdog | | `local_watchdog` | Опциональный локальный тик → stop при unclean exit | -| `llm_runtime` / `setup_wizard` | Opt-in Ollama/llama.cpp + `ollama-models.yaml` / `llamacpp-models.yaml` | +| `llm_runtime` / `setup_wizard` | Opt-in Ollama + `ollama-models.yaml` | | `idle_killer` / `hold` | systemd на VM + hold-файл | | `ready` / `snapshot` | Idle backend + boot snapshot | | `access_card` | URL / MCP panel после ready | diff --git a/docs/cli.md b/docs/cli.md index 292b19b..7eb0841 100644 --- a/docs/cli.md +++ b/docs/cli.md @@ -51,7 +51,7 @@ gpu-rent up --yes --ollama | Команда | Поведение | | --- | --- | -| `gpu-rent setup` | Wizard: манифесты (в т.ч. ollama/llamacpp-models), `LLM_RUNTIME` (нумерованное меню), пресет, опционально local-watchdog | +| `gpu-rent setup` | Wizard: манифесты (в т.ч. ollama-models), `LLM_RUNTIME` (нумерованное меню), пресет, опционально local-watchdog | | `gpu-rent doctor` | Preflight **без** create. Exit ≠ 0 → сессию начинать нельзя | | `gpu-rent flavors` | Скан `SCAN_POOLS` × `FLAVOR_PREFERENCE`, список в текущем регионе | | `gpu-rent dry-run` | План без mutating-вызовов | @@ -59,8 +59,8 @@ gpu-rent up --yes --ollama | `gpu-rent up --yes` | Без вопросов; flavor из `FLAVOR_PREFERENCE` / `DEFAULT_FLAVOR_ID` | | `up --keep-on-fail` | Не гасить GPU при ошибке install (по умолчанию `UP_STOP_ON_FAIL=true` → `stop`) | | `gpu-rent up -v` / `--verbose` | Полная таблица doctor на `up` (по умолчанию кратко) | -| `gpu-rent up --ollama` / `--llamacpp` / `--llm …` | LLM рядом со SwarmUI | -| `gpu-rent up --no-swarm` / `--llm-only` | Только LLM (нужен `--ollama`/`--llamacpp`); без clone SwarmUI | +| `gpu-rent up --ollama` / `--llm ollama` | Ollama рядом со SwarmUI | +| `gpu-rent up --no-swarm` / `--llm-only` | Только Ollama (нужен `--ollama` / `LLM_RUNTIME=ollama`); без clone SwarmUI | | `gpu-rent up --no-update` | Без `git pull` SwarmUI/extensions (только недостающие clone) | | `gpu-rent up --no-tunnel` | Только облако | | `gpu-rent up --no-spot` | Не preemptible | @@ -138,7 +138,7 @@ Exit 0 → можно `up`. Exit 1 → причина в таблице / кра | --- | --- | | `.env` | Секреты и OpenStack (`OS_*`, токены). Не в git | | `gpu-rent.vars` | Несекретные дефолты; читают лаунчеры и CLI. Пример: `gpu-rent.vars.example` | -| `models.yaml` / `extensions.yaml` / `ollama-models.yaml` / `llamacpp-models.yaml` | Манифесты (gitignore; из `*.example.yaml`) | +| `models.yaml` / `extensions.yaml` / `ollama-models.yaml` | Манифесты (gitignore; из `*.example.yaml`) | ### Лаунчер (`gpu-rent.vars`) @@ -192,7 +192,6 @@ AUTOCOMPLETE_ENABLED=true SWARMUI_LOCAL_PORT=17801 LLM_RUNTIME=none OLLAMA_LOCAL_PORT=17811 -LLAMACPP_LOCAL_PORT=17812 UPDATE_GIT=true DEFAULT_FLAVOR_ID= @@ -233,12 +232,12 @@ Application credential для idle-killer CLI создаёт на `up` (узко 1. Nova `ACTIVE` 2. TCP 22 / SSH 3. Backend Idle (если SwarmUI) → toast (если `NOTIFY_READY`) -4. На VM: HTTP сервисов стека (SwarmUI `:7801` / Ollama `:11434` / llama.cpp `:8080`) — `verify_stack_on_vm` +4. На VM: HTTP сервисов стека (SwarmUI `:7801` / Ollama `:11434`) — `verify_stack_on_vm` 5. На VM: **nvidia-smi / CUDA** (fail-fast) + **torch+cuda** в Comfy venv при SwarmUI (ждём) — `verify_gpu_env` 6. В логе: строка **`тайминг up:`** (SSH / bootstrap / Idle / verify / …) 7. Туннель + проверка **localhost** тех же сервисов → access-card (красная рамка, если killer failed) -`gpu-rent logs --unit swarm|ollama|llamacpp|killer` — фильтр journalctl. +`gpu-rent logs --unit swarm|ollama|killer` — фильтр journalctl. `gpu-rent status` — killer/hold, последний стек/GPU-env, тайминг up. Локальный порт UI: **17801** (на VM по-прежнему 7801 на loopback). diff --git a/docs/concept.md b/docs/concept.md index 900301c..ec2c1cd 100644 --- a/docs/concept.md +++ b/docs/concept.md @@ -24,7 +24,7 @@ GPU в облаке дорогой. Веса для генерации карт - Первый clone git-реп расширений SwarmUI и ComfyUI nodes из `extensions.yaml`. - Autocomplete: word-list в `Data/Autocompletions` до старта UI, на каждом `up` проверка новой версии. - `doctor` до create; фоллбек flavor; интерактивный выбор flavor/диска; `hold`; toast Idle; `open` на 17801. -- Opt-in LLM (Ollama / llama.cpp) через `setup` / флаги / `LLM_RUNTIME`; `capture` ссылок с VM. +- Opt-in LLM (Ollama) через `setup` / флаги / `LLM_RUNTIME`; `capture` ссылок с VM. - `status`: диск, окно preempt 24 ч, killer / LLM. - Snapshot boot после первого удачного bootstrap. diff --git a/docs/decisions.md b/docs/decisions.md index e429d03..a104b1a 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -10,7 +10,7 @@ | Civitai хост | Дефолт API `civitai.red` (полный каталог). `.com` — SFW-витрина, NSFW с неё часто 404. Ссылки `.com`/`.red`/`.green` в манифесте принимаем. 404 → один retry на второй хост. Один токен на оба домена | | Пул GPU | Перед `up`/`flavors` сканируем `SCAN_POOLS` (дефолт `ru-6,ru-7`). `ru-6` — мультизональный: ходим на `https://ru-6.cloud.api.selcloud.ru/compute/` тем же токеном (SDK-каталог часто знает только RC-пул). Собираем типы GPU из extra_specs и совпадения с `FLAVOR_PREFERENCE`. Автоматом `.env` не пишем — печатаем рекомендацию `OS_REGION_NAME` / `GPU_RENT_AZ` | | Манифест моделей | `/models.yaml`, типы: checkpoint / lora / vae / embedding / controlnet / upscaler. В git только `models.example.yaml` | -| Расширения | `/extensions.yaml`: git-репы `swarmui` → `src/Extensions`, `comfy` → DLNodes. Поле `requires: none\|ollama\|llamacpp\|any-llm` фильтрует по `LLM_RUNTIME`. Пример: swarm-assistent с `requires: ollama` | +| Расширения | `/extensions.yaml`: git-репы `swarmui` → `src/Extensions`, `comfy` → DLNodes. Поле `requires: none\|ollama\|any-llm` фильтрует по `LLM_RUNTIME`. Пример: swarm-assistent с `requires: ollama` | | Autocomplete | До первого старта: скачать word-list в `Data/Autocompletions`, прописать `DefaultUser.AutoComplete.Source`. На каждом `up` сверить GitHub blob sha и обновить файл, если изменился. Дефолт: `tags/danbooru.csv` из a1111-sd-webui-tagcomplete (как в доке SwarmUI) | | Доступ | Браузер на туннеле; MCP переключается на облако, пока оно живо; HTTP API SwarmUI через тот же туннель | | Preemptible | По умолчанию всегда. Обычный сервер — только `--no-spot` | @@ -22,7 +22,7 @@ | `up` / `tunnel` | `up` по умолчанию после ready открывает туннель `:17801`, печатает URL и ждёт. `--no-tunnel` — только облако. Ctrl+C на туннеле GPU не гасит (`stop` отдельно). Команда `tunnel` остаётся для повторного входа | | Interactive `up` | Без `--yes`: нумерованные меню (LLM, пресет, flavor #, data GB, preemptible) → confirm. Изменения можно сохранить в `gpu-rent.vars` | | Git update | На каждом `up` по умолчанию: `git pull` SwarmUI + репы из `extensions.yaml` + уже установленные на data (`Extensions`/`DLNodes`). `--no-update` или `UPDATE_GIT=false` — не тянуть | -| LLM (opt-in) | `none` по умолчанию. Флаги / `LLM_RUNTIME` / меню. Манифесты: `ollama-models.yaml`, `llamacpp-models.yaml`. Порты 17811 / 17812. Prompt-help, не замена SwarmUI | +| LLM (opt-in) | `none` по умолчанию. Флаги / `LLM_RUNTIME` / меню. Манифест: `ollama-models.yaml`. Порт 17811. Prompt-help, не замена SwarmUI | | llm-only | `ENABLE_SWARMUI=false` / `WORKLOAD=llm` / `--no-swarm`/`--llm-only`: GPU + LLM без SwarmUI (bootstrap только data disk) | | Capture | `gpu-rent capture` — инвентарь VM → merge **ссылок** в локальные yaml (веса не качать) | | Perf auto-tune | На `up`: probe GPU → tier. Swarm/Comfy: sageattention ExtraArgs на Ampere+ ≥16 GiB. Ollama: flash/KV/keep-alive + `GPU_OVERHEAD` чтобы оставить VRAM под Krea | diff --git a/docs/extensions.md b/docs/extensions.md index 1f12be6..b6cf5c2 100644 --- a/docs/extensions.md +++ b/docs/extensions.md @@ -38,7 +38,7 @@ swarmui: - url: https://gitea.hsrv.site/mrleo1nid/swarm-assistent.git ref: main dir: swarm-assistent - requires: ollama # none | ollama | llamacpp | any-llm; default none = always + requires: ollama # none | ollama | any-llm; default none = always - url: https://github.com/example/SwarmUI-SomeExt.git ref: main dir: SomeExt @@ -51,8 +51,7 @@ comfy: | --- | --- | | `none` (или поле отсутствует) | всегда | | `ollama` | только при `LLM_RUNTIME=ollama` | -| `llamacpp` | только при `LLM_RUNTIME=llamacpp` | -| `any-llm` | при `ollama` или `llamacpp` | +| `any-llm` | при `LLM_RUNTIME=ollama` | Строки с несовпавшим `requires` пропускаются (лог), остальные ставятся как обычно. В `extensions.example.yaml` по умолчанию — **swarm-assistent** с `requires: ollama` (чат/vision под Krea 2). diff --git a/docs/llm.md b/docs/llm.md index 98f246a..b8eccf3 100644 --- a/docs/llm.md +++ b/docs/llm.md @@ -1,6 +1,6 @@ # LLM рядом со SwarmUI (opt-in) -По умолчанию поднимается **только SwarmUI**. Ollama или llama.cpp — отдельно (помощь с промптами и т.п.). +По умолчанию поднимается **только SwarmUI**. Ollama — отдельно (помощь с промптами и т.п.). Интерактивные вопросы — **нумерованные меню** (Enter = вариант со ←). Ключ словом (`recommended`, `keep`) тоже принимается. @@ -9,10 +9,10 @@ | # | стек | смысл | | --- | --- | --- | | 1 | SwarmUI | только генерация картинок | -| 2 | SwarmUI + LLM | UI + Ollama/llama.cpp на той же карте | -| 3 | только LLM | **без** SwarmUI — Ollama или llama.cpp | +| 2 | SwarmUI + LLM | UI + Ollama на той же карте | +| 3 | только LLM | **без** SwarmUI — только Ollama | -Флаги: `up --yes --llm-only --llamacpp` / `--no-swarm --ollama`. Vars: `ENABLE_SWARMUI=false` или `WORKLOAD=llm`. +Флаги: `up --yes --llm-only --ollama` / `--no-swarm --ollama`. Vars: `ENABLE_SWARMUI=false` или `WORKLOAD=llm`. --- @@ -30,8 +30,8 @@ gpu-rent setup ```text gpu-rent up --yes --ollama -gpu-rent up --yes --llm llamacpp -gpu-rent up --yes --llm-only --llamacpp # без SwarmUI +gpu-rent up --yes --llm ollama +gpu-rent up --yes --llm-only --ollama # без SwarmUI ``` С `--yes` пресеты не спрашивает (берёт существующий yaml / example). @@ -44,7 +44,7 @@ LLM_RUNTIME=ollama Потом `gpu-rent up --yes`. -Выключить: `LLM_RUNTIME=none` (на `up` старые unit’ы `gpu-rent-ollama` / `gpu-rent-llamacpp` останавливаются). +Выключить: `LLM_RUNTIME=none` (на `up` старые LLM unit’ы на VM останавливаются). Голый `gpu-rent` без args — **help**. Двойной клик: `GPU_RENT_DEFAULT_ARGS=up --yes`. Полный doctor на up: `up -v`. @@ -56,7 +56,6 @@ LLM_RUNTIME=ollama | --- | --- | --- | | SwarmUI | 7801 | **17801** | | Ollama | 11434 | **17811** | -| llama.cpp | 8080 | **17812** | ```text gpu-rent tunnel @@ -111,61 +110,12 @@ Unit `gpu-rent-ollama` читает `/mnt/swarm_data/.gpu-rent-gpu.json`: --- -## llama.cpp - -| Файл | Роль | -| --- | --- | -| `llamacpp-models.example.yaml` | шаблон | -| `llamacpp-models.yaml` | HTTPS URL на `.gguf` (gitignore) | - -На `up`: скачать GGUF → `/mnt/swarm_data/llamacpp/models` → `llama-server` + systemd. Уже скачанные крупные файлы не трогает. - -**Бинарник:** в GitHub Releases **нет** Linux CUDA — только Windows CUDA + Ubuntu CPU/Vulkan. Драйвер/CUDA runtime на VM ≠ готовый `llama-server` с CUDA: его нужно **собрать** (`nvcc`) или взять Vulkan prebuilt. - -Порядок по умолчанию (`LLAMACPP_BACKEND=auto`): - -1. Если на VM уже есть `nvcc` (часто после прошлого `up`) → **CUDA-сборка** (тихо + heartbeat 5–15 мин). -2. Иначе → Ubuntu **Vulkan** prebuilt (~30 MB). -3. Fallback на другой путь при ошибке. - -Был только Vulkan-stamp, а `nvcc` появился — следующий `up` сам пересоберёт CUDA. - -| Var | Зачем | -| --- | --- | -| `LLAMACPP_BACKEND=auto\|cuda\|vulkan` | выбор пути (default auto) | -| `LLAMACPP_TAG` | pin release (`b10545`) | -| `LLAMACPP_ASSET_URL` + `LLAMACPP_SHA256` | свой архив | -| `LLAMACPP_BUILD_CUDA=1` | форс CUDA; `=0` — никогда не собирать | -| `LLAMACPP_FORCE_REINSTALL=1` | снести бинарь и поставить заново | -| `LLAMACPP_NGL` / `LLAMACPP_CTX` | override GPU layers / context | -| `LLAMACPP_HOST` / `LLAMACPP_PORT` | bind (default `127.0.0.1:8080`) | -| `LLAMACPP_EXTRA_ARGS` | доп. флаги `llama-server` | -| `OLLAMA_VERSION` / `OLLAMA_SHA256` | pin Ollama | - -Готовые кейсы A–H — в `gpu-rent.vars.example`. - -### Пресеты (меню) - -| # | ключ | что | -| --- | --- | --- | -| 1 | **recommended** | Qwen2.5-VL 7B abliterate + mmproj (~4.7+0.8 GB, картинки+RU) | -| 2 | light | Qwen2.5 3B Instruct text Q4_K_M | -| 3 | text | Qwen2.5 7B abliterate text-only | -| 4 | stock | официальный 7B Instruct Q4_K_M | -| 5 | empty | только runtime | -| — | keep | не трогать yaml | - -`-ngl` / `-c` — по GPU probe. Для recommended abliterate GGUF нужен **`HF_TOKEN`** в `.env` (иначе 401). Вручную: положи GGUF в models и `systemctl restart gpu-rent-llamacpp`. - ---- - ## Idle-killer и LLM Busy (не гасить GPU): - `ollama pull` (маркер младше ~45 мин; старше сбрасывается); -- загруженная модель в Ollama; -- llama.cpp: слоты заняты. +- загруженная модель в Ollama. Ошибка установки LLM на `up` — **fail** (не тихий лог). @@ -178,9 +128,6 @@ Busy (не гасить GPU): ```bash OLLAMA_VERSION=0.6.5 OLLAMA_SHA256= -LLAMACPP_TAG=b10545 -# или LLAMACPP_ASSET_URL=... + LLAMACPP_SHA256=... -# CUDA из исходников (медленно): LLAMACPP_BUILD_CUDA=1 -# Переустановка бинаря: LLAMACPP_FORCE_REINSTALL=1 -# VRAM: LLAMACPP_NGL=40 LLAMACPP_CTX=4096 LLAMACPP_EXTRA_ARGS=--flash-attn on ``` + +Готовые кейсы — в `gpu-rent.vars.example`. diff --git a/docs/models.md b/docs/models.md index 49e5a25..03a5866 100644 --- a/docs/models.md +++ b/docs/models.md @@ -196,7 +196,7 @@ LOCAL_MODELS_DIR= # пусто = <корень приложения>/Models ## Hugging Face (резерв) -- **Скачивание:** `HF_TOKEN` в `.env` → GGUF (llama.cpp) и строки `models.yaml` с `huggingface.co/…/resolve/…`. +- **Скачивание:** `HF_TOKEN` в `.env` → строки `models.yaml` с `huggingface.co/…/resolve/…`. - **Capture:** если Civitai by-hash не нашёл файл → поиск на Hub по имени + LFS sha256 → URL в манифест. - **SwarmUI:** тот же токен прокидывается как `huggingface_api` (Model Downloader). - Abliterated / gated репозитории без токена почти всегда дают **401**. diff --git a/docs/setup.md b/docs/setup.md index 413ab8a..3189a90 100644 --- a/docs/setup.md +++ b/docs/setup.md @@ -49,7 +49,7 @@ chmod +x gpu-rent.sh - `env.example` → `.env` - `models.example.yaml` → `models.yaml` - `extensions.example.yaml` → `extensions.yaml` -- `ollama-models.example.yaml` / `llamacpp-models.example.yaml` → соответствующие yaml (лаунчер / setup) +- `ollama-models.example.yaml` → `ollama-models.yaml` (лаунчер / setup) - `gpu-rent.vars.example` → `gpu-rent.vars` Секреты только в `/.env` и runtime в `/.gpu-rent/` (оба в `.gitignore`). @@ -243,7 +243,7 @@ copy models.example.yaml models.yaml | --- | --- | | `--no-tunnel` | только облако; UI потом: `tunnel --open` | | `--no-update` | не `git pull` SwarmUI/extensions | -| `--ollama` / `--llamacpp` | LLM рядом ([llm.md](llm.md)) | +| `--ollama` / `--llm ollama` | LLM рядом ([llm.md](llm.md)) | | `--no-swarm` / `--llm-only` | только LLM, без SwarmUI (нужен runtime) | | `--no-spot` | обычный (не preemptible) тариф | | `--flavor ID` | явный flavor (пропускает меню flavor) | diff --git a/docs/spike-notes.md b/docs/spike-notes.md index a6c524c..390f71c 100644 --- a/docs/spike-notes.md +++ b/docs/spike-notes.md @@ -108,7 +108,7 @@ | Шаг | OK? | | --- | --- | -| Ollama :17811 или llama.cpp :17812 через туннель | | +| Ollama :17811 через туннель | | | pull / GGUF из манифеста | | | `LLM_RUNTIME=none` гасит unit на следующем up | | diff --git a/docs/swarmui.md b/docs/swarmui.md index bdfb9af..929ff56 100644 --- a/docs/swarmui.md +++ b/docs/swarmui.md @@ -20,7 +20,7 @@ 6. Если есть Civitai-токен и манифест моделей: seed весов **до** первого старта SwarmUI; иначе — дефолтная модель установщика. 7. Push непустых `Models/` / `Wildcards/` / `CustomWorkflows/`. 8. systemd unit `swarmui`: `./launch-linux.sh --launch_mode none --host 127.0.0.1 --port 7801`. -9. **GPU probe** → `/mnt/swarm_data/.gpu-rent-gpu.json` (VRAM / compute cap / tier) — до старта UI; Ollama/llama.cpp читают его при install. +9. **GPU probe** → `/mnt/swarm_data/.gpu-rent-gpu.json` (VRAM / compute cap / tier) — до старта UI; Ollama читает его при install. 10. Старт SwarmUI → seed LLM → idle-killer → **wait backend Idle**. 11. **Perf tune после Idle** — `triton`+`sageattention` в Comfy venv и `--use-sage-attention` в `Data/Backends.fds` (маркер `.gpu-rent-perf-tuned`; повтор при смене GPU). 12. Авторизация SwarmUI включена, токен в `Data` на диске. diff --git a/env.example b/env.example index c17ec3d..fe11e41 100644 --- a/env.example +++ b/env.example @@ -38,29 +38,18 @@ AUTOCOMPLETE_GITHUB_REF=main AUTOCOMPLETE_FILENAME=danbooru.csv SWARMUI_LOCAL_PORT=17801 -# Optional LLM beside SwarmUI: none | ollama | llamacpp (setup / --ollama / --llamacpp) +# Optional LLM beside SwarmUI: none | ollama (setup / --ollama / --llm ollama) # ENABLE_SWARMUI=true -# WORKLOAD=llm # llm-only (no SwarmUI); requires LLM_RUNTIME≠none +# WORKLOAD=llm # llm-only (no SwarmUI); requires LLM_RUNTIME=ollama LLM_RUNTIME=none OLLAMA_LOCAL_PORT=17811 -LLAMACPP_LOCAL_PORT=17812 # OLLAMA_MODELS_MANIFEST= -# LLAMACPP_MODELS_MANIFEST= -# Pin / тонкая настройка LLM (несecреты; удобнее в gpu-rent.vars — см. кейсы A–H): -# LLAMACPP_TAG=b10545 -# LLAMACPP_BACKEND=auto -# LLAMACPP_BUILD_CUDA=1 -# LLAMACPP_FORCE_REINSTALL=1 -# LLAMACPP_ASSET_URL= -# LLAMACPP_SHA256= -# LLAMACPP_NGL=40 -# LLAMACPP_CTX=4096 -# LLAMACPP_EXTRA_ARGS=--flash-attn on +# Pin Ollama (несecреты; удобнее в gpu-rent.vars): # OLLAMA_VERSION=0.6.5 # OLLAMA_SHA256= # CIVITAI_API_TOKEN= # CIVITAI_API_HOST=civitai.red -# Hugging Face (GGUF / gated HF URLs / capture fallback metadata): +# Hugging Face (HF URLs in models.yaml / capture fallback metadata): # HF_TOKEN= # or HUGGING_FACE_HUB_TOKEN — https://huggingface.co/settings/tokens # git pull SwarmUI + extensions on each up (default true). CLI: --no-update UPDATE_GIT=true diff --git a/extensions.example.yaml b/extensions.example.yaml index 1083cef..b1fabf6 100644 --- a/extensions.example.yaml +++ b/extensions.example.yaml @@ -2,8 +2,8 @@ # Empty/missing file → no extra extensions, stock SwarmUI. # swarmui = C# repos cloned to src/Extensions # comfy = Python custom nodes cloned to ComfyUI DLNodes -# requires: none (default) | ollama | llamacpp | any-llm -# — clone only when LLM_RUNTIME matches (ollama / llamacpp / either) +# requires: none (default) | ollama | any-llm +# — clone only when LLM_RUNTIME matches (ollama / either LLM) swarmui: - url: https://gitea.hsrv.site/mrleo1nid/swarm-assistent.git diff --git a/gpu-rent.ps1 b/gpu-rent.ps1 index 775d99f..0997229 100644 --- a/gpu-rent.ps1 +++ b/gpu-rent.ps1 @@ -119,7 +119,6 @@ function Copy-IfMissing { Copy-IfMissing (Join-Path $Root "models.example.yaml") (Join-Path $Root "models.yaml") "models.yaml" Copy-IfMissing (Join-Path $Root "extensions.example.yaml") (Join-Path $Root "extensions.yaml") "extensions.yaml" Copy-IfMissing (Join-Path $Root "ollama-models.example.yaml") (Join-Path $Root "ollama-models.yaml") "ollama-models.yaml" -Copy-IfMissing (Join-Path $Root "llamacpp-models.example.yaml") (Join-Path $Root "llamacpp-models.yaml") "llamacpp-models.yaml" Copy-IfMissing (Join-Path $Root "gpu-rent.vars.example") (Join-Path $Root "gpu-rent.vars") "gpu-rent.vars" Import-GpuRentVars (Join-Path $Root "gpu-rent.vars") diff --git a/gpu-rent.sh b/gpu-rent.sh index ce1bf42..0f9915f 100644 --- a/gpu-rent.sh +++ b/gpu-rent.sh @@ -99,10 +99,6 @@ if [[ ! -f "$ROOT/ollama-models.yaml" && -f "$ROOT/ollama-models.example.yaml" ] cp "$ROOT/ollama-models.example.yaml" "$ROOT/ollama-models.yaml" echo "gpu-rent: created ollama-models.yaml" fi -if [[ ! -f "$ROOT/llamacpp-models.yaml" && -f "$ROOT/llamacpp-models.example.yaml" ]]; then - cp "$ROOT/llamacpp-models.example.yaml" "$ROOT/llamacpp-models.yaml" - echo "gpu-rent: created llamacpp-models.yaml" -fi if [[ ! -f "$ROOT/gpu-rent.vars" && -f "$ROOT/gpu-rent.vars.example" ]]; then cp "$ROOT/gpu-rent.vars.example" "$ROOT/gpu-rent.vars" echo "gpu-rent: created gpu-rent.vars" diff --git a/gpu-rent.vars.example b/gpu-rent.vars.example index 38855b8..69b2b8e 100644 --- a/gpu-rent.vars.example +++ b/gpu-rent.vars.example @@ -31,59 +31,24 @@ # Подробнее: docs/llm.md # --------------------------------------------------------------------------- -# --- A) Default: SwarmUI + llama.cpp (auto: CUDA если nvcc на VM, иначе Vulkan) --- -# LLM_RUNTIME=llamacpp +# --- A) Default: SwarmUI + Ollama --- +# LLM_RUNTIME=ollama # ENABLE_SWARMUI=true -# LLAMACPP_TAG=b10545 -# LLAMACPP_BACKEND=auto +# OLLAMA_LOCAL_PORT=17811 # --- B) Только LLM, без SwarmUI --- # WORKLOAD=llm -# LLM_RUNTIME=llamacpp -# LLAMACPP_TAG=b10545 +# LLM_RUNTIME=ollama -# --- C) Явно CUDA (и снести старый Vulkan-бинарь) --- -# LLM_RUNTIME=llamacpp -# LLAMACPP_TAG=b10545 -# LLAMACPP_BACKEND=cuda -# LLAMACPP_FORCE_REINSTALL=1 - -# --- C2) Явно быстрый Vulkan, без compile --- -# LLM_RUNTIME=llamacpp -# LLAMACPP_BACKEND=vulkan -# LLAMACPP_BUILD_CUDA=0 - -# --- D) Свой бинарь / pin URL (supply-chain) --- -# LLM_RUNTIME=llamacpp -# LLAMACPP_ASSET_URL=https://github.com/ggml-org/llama.cpp/releases/download/b10545/llama-b10545-bin-ubuntu-vulkan-x64.tar.gz -# LLAMACPP_SHA256= -# LLAMACPP_FORCE_REINSTALL=1 - -# --- E) Делим 4090 со SwarmUI: меньше слоёв / короче контекст --- -# LLM_RUNTIME=llamacpp -# LLAMACPP_NGL=40 -# LLAMACPP_CTX=4096 -# LLAMACPP_EXTRA_ARGS=--flash-attn on - -# --- F) Длинный контекст / почти весь VRAM под LLM (llm-only) --- -# WORKLOAD=llm -# LLM_RUNTIME=llamacpp -# LLAMACPP_NGL=99 -# LLAMACPP_CTX=32768 -# LLAMACPP_EXTRA_ARGS=--parallel 1 - -# --- G) Ollama вместо llama.cpp + pin версии --- +# --- C) Pin версии Ollama (supply-chain) --- # LLM_RUNTIME=ollama # OLLAMA_VERSION=0.6.5 # OLLAMA_SHA256= -# OLLAMA_LOCAL_PORT=17811 -# --- H) Выключить LLM (гасит unit’ы на следующем up) --- +# --- D) Выключить LLM (гасит unit’ы на следующем up) --- # LLM_RUNTIME=none -# Порты туннеля (localhost): +# Порт туннеля (localhost): # OLLAMA_LOCAL_PORT=17811 -# LLAMACPP_LOCAL_PORT=17812 -# Свой путь к манифесту GGUF/pull: -# LLAMACPP_MODELS_MANIFEST= +# Свой путь к манифесту pull: # OLLAMA_MODELS_MANIFEST= diff --git a/llamacpp-models.example.yaml b/llamacpp-models.example.yaml deleted file mode 100644 index 8aab03f..0000000 --- a/llamacpp-models.example.yaml +++ /dev/null @@ -1,16 +0,0 @@ -# Copy to llamacpp-models.yaml (gitignored). Used when LLM_RUNTIME=llamacpp. -# url = direct HTTPS link to a .gguf; mmproj = vision projector (Qwen2.5-VL). -# Empty models: [] → only llama-server, GGUF клади вручную. -# Purpose: prompt-help / vision beside SwarmUI (RU/EN, low refusal). - -models: - # Recommended: Qwen2.5-VL 7B abliterate Q4_K_M + mmproj (~4.7 + 0.8 GB) - - url: https://huggingface.co/mradermacher/Qwen2.5-VL-7B-Instruct-abliterated-GGUF/resolve/main/Qwen2.5-VL-7B-Instruct-abliterated.Q4_K_M.gguf - mmproj: https://huggingface.co/mradermacher/Qwen2.5-VL-7B-Instruct-abliterated-GGUF/resolve/main/Qwen2.5-VL-7B-Instruct-abliterated.mmproj-Q8_0.gguf - default: true - - # Text-only abliterate 7B (~4.7 GB): - # - url: https://huggingface.co/RichardErkhov/huihui-ai_-_Qwen2.5-7B-Instruct-abliterated-gguf/resolve/main/Qwen2.5-7B-Instruct-abliterated.Q4_K_M.gguf - - # Lighter text (~2 GB): - # - url: https://huggingface.co/bartowski/Qwen2.5-3B-Instruct-GGUF/resolve/main/Qwen2.5-3B-Instruct-Q4_K_M.gguf diff --git a/ollama-models.example.yaml b/ollama-models.example.yaml index edc825e..71356c4 100644 --- a/ollama-models.example.yaml +++ b/ollama-models.example.yaml @@ -1,17 +1,15 @@ # Copy to ollama-models.yaml (gitignored). Used when LLM_RUNTIME=ollama. -# name = exact tag for `ollama pull`. Empty models: [] → runtime only, no pull. -# Purpose: SwarmUI prompt help + vision (RU/EN, low refusal). +# name = exact tag for `ollama pull` (Ollama library / community). +# Empty models: [] → runtime only, no pull. +# Purpose: SwarmUI prompt help — vision + RU/EN on the same GPU as diffusion. models: - # Recommended: vision + Russian/English, ~6GB, abliterated + # Recommended (~6GB): vision + Russian/English, low refusal - name: huihui_ai/qwen2.5-vl-abliterated:7b default: true - # Lighter vision: - # - name: huihui_ai/qwen2.5-vl-abliterated:3b - - # Text-only abliterate (~5GB): - # - name: huihui_ai/qwen2.5-abliterate:7b - - # Official stock (more refusals): - # - name: qwen2.5:7b + # Alternatives (uncomment / use presets on setup|up): + # light — huihui_ai/qwen2.5-vl-abliterated:3b (~3GB) + # stock — qwen2.5vl:7b (official library) + # text — huihui_ai/qwen2.5-abliterate:7b (no vision, ~5GB) + # big — qwen2.5vl:32b (~21GB; llm-only) diff --git a/src/gpu_rent/access_card.py b/src/gpu_rent/access_card.py index ce88c86..9c47e6d 100644 --- a/src/gpu_rent/access_card.py +++ b/src/gpu_rent/access_card.py @@ -71,19 +71,6 @@ def collect_access_links(cfg: Config, *, tunneled: bool) -> list[AccessLink]: ), ] ) - elif runtime == "llamacpp": - p = cfg.llamacpp_local_port - links.extend( - [ - AccessLink("llama.cpp", f"http://127.0.0.1:{p}", "OpenAI-compatible"), - AccessLink( - "OpenAI /v1", - f"http://127.0.0.1:{p}/v1/chat/completions", - "chat completions", - ), - AccessLink("Models", f"http://127.0.0.1:{p}/v1/models", "list"), - ] - ) if not links: links.append( AccessLink( @@ -138,7 +125,7 @@ def render_access_panel( cmds.add_column(style="dim", no_wrap=True) cmds.add_column() cmds.add_row("открыть UI", "gpu-rent open") - if resolve_llm_runtime(cfg) in {"ollama", "llamacpp"}: + if resolve_llm_runtime(cfg) == "ollama": cmds.add_row("открыть LLM", "gpu-rent open --llm") cmds.add_row("hold killer", "gpu-rent hold") cmds.add_row("стоп GPU", "gpu-rent stop") diff --git a/src/gpu_rent/cli.py b/src/gpu_rent/cli.py index 7abb046..a326191 100644 --- a/src/gpu_rent/cli.py +++ b/src/gpu_rent/cli.py @@ -335,7 +335,7 @@ def status() -> None: rt = resolve_llm_runtime(cfg) noted = notes.get("llm_runtime") llm_err = notes.get("llm_error") - detail = f"{rt}; ollama :{cfg.ollama_local_port} / llamacpp :{cfg.llamacpp_local_port}" + detail = f"{rt}; ollama :{cfg.ollama_local_port}" if noted and noted != rt: detail += f" (notes: {noted})" if llm_err: @@ -370,7 +370,7 @@ def status() -> None: def open( llm: bool = typer.Option(False, "--llm", help="Открыть LLM API URL вместо SwarmUI"), ) -> None: - """Открыть браузер на SwarmUI :17801 (или --llm / llm-only на Ollama/llama.cpp).""" + """Открыть браузер на SwarmUI :17801 (или --llm / llm-only на Ollama).""" cfg = load_config(require_auth=False) use_llm = llm or not bool(getattr(cfg, "enable_swarmui", True)) if use_llm: @@ -379,8 +379,6 @@ def open( runtime = resolve_llm_runtime(cfg) if runtime == "ollama": port = cfg.ollama_local_port - elif runtime == "llamacpp": - port = cfg.llamacpp_local_port else: console.print("[red]LLM не выбран[/red] (LLM_RUNTIME / gpu-rent setup)") raise typer.Exit(1) @@ -398,9 +396,9 @@ def open( @app.command() def setup( - llm: Optional[str] = typer.Option(None, "--llm", help="none|ollama|llamacpp"), + llm: Optional[str] = typer.Option(None, "--llm", help="none|ollama"), ollama_preset: Optional[str] = typer.Option( - None, "--ollama-preset", help="recommended|light|stock|alt|empty" + None, "--ollama-preset", help="recommended|light|stock|text|big|empty" ), watchdog: Optional[bool] = typer.Option( None, "--watchdog/--no-watchdog", help="Поставить local-watchdog" @@ -466,10 +464,9 @@ def up( help="Полный doctor-таблица на up (по умолчанию кратко)", ), llm: Optional[str] = typer.Option( - None, "--llm", help="none|ollama|llamacpp (override LLM_RUNTIME)" + None, "--llm", help="none|ollama (override LLM_RUNTIME)" ), ollama: bool = typer.Option(False, "--ollama", help="То же что --llm ollama"), - llamacpp: bool = typer.Option(False, "--llamacpp", help="То же что --llm llamacpp"), no_swarm: bool = typer.Option( False, "--no-swarm", @@ -489,7 +486,7 @@ def up( write_ollama_models_preset, ) from gpu_rent.paths import vars_path - from gpu_rent.prompts import MenuItem, prompt_menu + from gpu_rent.prompts import prompt_menu from gpu_rent.timing import clock_elapsed, clock_reset, format_duration from gpu_rent.varsfile import upsert_vars @@ -506,7 +503,6 @@ def up( runtime = decide_runtime( flag=llm, ollama_flag=ollama, - llamacpp_flag=llamacpp, from_config=cfg.llm_runtime, ) except ValueError as exc: @@ -514,11 +510,10 @@ def up( enable_swarm = False if no_swarm else cfg.enable_swarmui asked_model_preset = False - llm_flags = bool(llm or ollama or llamacpp or no_swarm) + llm_flags = bool(llm or ollama or no_swarm) if not yes and not llm_flags: from gpu_rent.llm_runtime import ( - llamacpp_preset_menu, ollama_preset_menu, workload_menu, ) @@ -548,47 +543,11 @@ def up( elif stack == "both": enable_swarm = True if runtime == "none": - try: - choice = prompt_menu( - "LLM runtime", - [ - MenuItem("ollama", "Ollama (+ pull моделей)"), - MenuItem("llamacpp", "llama.cpp server (+ GGUF)"), - ], - default="ollama", - ask=_ask, - show=log, - ) - runtime = decide_runtime( - flag=choice, - ollama_flag=False, - llamacpp_flag=False, - from_config="none", - ) - except ValueError as exc: - raise GpuRentError(str(exc)) from exc + runtime = "ollama" else: enable_swarm = False if runtime == "none": - try: - choice = prompt_menu( - "LLM runtime", - [ - MenuItem("ollama", "Ollama (+ pull моделей)"), - MenuItem("llamacpp", "llama.cpp server (+ GGUF)"), - ], - default="llamacpp", - ask=_ask, - show=log, - ) - runtime = decide_runtime( - flag=choice, - ollama_flag=False, - llamacpp_flag=False, - from_config="none", - ) - except ValueError as exc: - raise GpuRentError(str(exc)) from exc + runtime = "ollama" if typer.confirm("Запомнить стек в gpu-rent.vars?", default=True): upsert_vars( @@ -614,68 +573,31 @@ def up( if preset not in {"keep", "example"}: write_ollama_models_preset(cfg.ollama_models_manifest, preset) asked_model_preset = True - elif runtime == "llamacpp": - from gpu_rent.llm_runtime import ( - ensure_llamacpp_manifest_from_example, - write_llamacpp_models_preset, - ) - - ensure_llamacpp_manifest_from_example() - try: - preset = prompt_menu( - "llama.cpp GGUF", - llamacpp_preset_menu(include_keep=False), - default="recommended", - ask=_ask, - show=log, - ) - except ValueError as exc: - raise GpuRentError(str(exc)) from exc - if preset not in {"keep", "example"}: - write_llamacpp_models_preset(cfg.llamacpp_models_manifest, preset) - asked_model_preset = True # Runtime уже в vars — спросить пресет, default=keep. - if not yes and not asked_model_preset and runtime in {"llamacpp", "ollama"}: - from gpu_rent.llm_runtime import ( - ensure_llamacpp_manifest_from_example, - llamacpp_preset_menu, - ollama_preset_menu, - write_llamacpp_models_preset, - ) + if not yes and not asked_model_preset and runtime == "ollama": + from gpu_rent.llm_runtime import ollama_preset_menu def _ask2(msg: str, default: str = "") -> str: return typer.prompt(msg, default=default) try: - if runtime == "llamacpp": - ensure_llamacpp_manifest_from_example() - key = prompt_menu( - "llama.cpp GGUF", - llamacpp_preset_menu(include_keep=True), - default="keep", - ask=_ask2, - show=log, - ) - if key not in {"keep", "example", ""}: - write_llamacpp_models_preset(cfg.llamacpp_models_manifest, key) - else: - ensure_ollama_manifest_from_example() - key = prompt_menu( - "Ollama preset", - ollama_preset_menu(include_keep=True), - default="keep", - ask=_ask2, - show=log, - ) - if key not in {"keep", "example", ""}: - write_ollama_models_preset(cfg.ollama_models_manifest, key) + ensure_ollama_manifest_from_example() + key = prompt_menu( + "Ollama preset", + ollama_preset_menu(include_keep=True), + default="keep", + ask=_ask2, + show=log, + ) + if key not in {"keep", "example", ""}: + write_ollama_models_preset(cfg.ollama_models_manifest, key) except ValueError as exc: raise GpuRentError(str(exc)) from exc if not enable_swarm and runtime == "none": raise GpuRentError( - "llm-only требует --ollama / --llamacpp / --llm … " + "llm-only требует --ollama / --llm ollama " "(или убери --no-swarm / ENABLE_SWARMUI=true)" ) @@ -796,7 +718,7 @@ def logs( None, "--unit", "-u", - help="swarm|ollama|llamacpp|killer|cloud-init (по умолчанию — всё)", + help="swarm|ollama|killer|cloud-init (по умолчанию — всё)", ), lines: int = typer.Option(80, "--lines", "-n", help="Строк journalctl"), ) -> None: @@ -812,8 +734,6 @@ def logs( "swarm": "swarmui", "swarmui": "swarmui", "ollama": "ollama", - "llamacpp": "llamacpp", - "llama": "llamacpp", "killer": "gpu-rent-idle-killer", "idle-killer": "gpu-rent-idle-killer", "idle": "gpu-rent-idle-killer", @@ -823,7 +743,7 @@ def logs( if key not in aliases: raise GpuRentError( f"неизвестный --unit={unit!r}; " - "ожидаю: swarm|ollama|llamacpp|killer|cloud-init|all" + "ожидаю: swarm|ollama|killer|cloud-init|all" ) target = aliases[key] n = max(10, min(int(lines), 500)) @@ -835,7 +755,7 @@ def logs( ) journal_units = [] if target == "all": - journal_units = ["swarmui", "ollama", "llamacpp", "gpu-rent-idle-killer"] + journal_units = ["swarmui", "ollama", "gpu-rent-idle-killer"] elif target != "cloud-init": journal_units = [target] for ju in journal_units: diff --git a/src/gpu_rent/config.py b/src/gpu_rent/config.py index 3e9610d..c443f47 100644 --- a/src/gpu_rent/config.py +++ b/src/gpu_rent/config.py @@ -14,7 +14,6 @@ from gpu_rent.paths import ( default_ssh_key_path, env_path, extensions_manifest_path, - llamacpp_models_manifest_path, migrate_legacy_if_needed, models_manifest_path, ollama_models_manifest_path, @@ -107,9 +106,7 @@ class Config: llm_runtime: str enable_swarmui: bool ollama_models_manifest: Path - llamacpp_models_manifest: Path ollama_local_port: int - llamacpp_local_port: int default_flavor_id: str flavor_preference: tuple[str, ...] @@ -185,10 +182,6 @@ def load_config(*, require_auth: bool = True) -> Config: (os.environ.get("OLLAMA_MODELS_MANIFEST") or "").strip() or str(ollama_models_manifest_path()) ).expanduser() - llamacpp_manifest = Path( - (os.environ.get("LLAMACPP_MODELS_MANIFEST") or "").strip() - or str(llamacpp_models_manifest_path()) - ).expanduser() try: llm_runtime = normalize_runtime(os.environ.get("LLM_RUNTIME")) @@ -243,9 +236,7 @@ def load_config(*, require_auth: bool = True) -> Config: llm_runtime=llm_runtime, enable_swarmui=enable_swarmui, ollama_models_manifest=ollama_manifest, - llamacpp_models_manifest=llamacpp_manifest, ollama_local_port=_as_int(os.environ.get("OLLAMA_LOCAL_PORT"), 17811), - llamacpp_local_port=_as_int(os.environ.get("LLAMACPP_LOCAL_PORT"), 17812), default_flavor_id=(os.environ.get("DEFAULT_FLAVOR_ID") or "").strip(), flavor_preference=_csv( os.environ.get("FLAVOR_PREFERENCE"), diff --git a/src/gpu_rent/doctor.py b/src/gpu_rent/doctor.py index cda197e..9cdba17 100644 --- a/src/gpu_rent/doctor.py +++ b/src/gpu_rent/doctor.py @@ -282,7 +282,6 @@ def _civitai(cfg: Config, checks: list[Check]) -> None: def _huggingface(cfg: Config, checks: list[Check]) -> None: from gpu_rent.huggingface import is_huggingface_url, probe_whoami - from gpu_rent.llm_runtime import normalize_runtime, parse_llamacpp_models from gpu_rent.manifests import parse_models needs_hf = False @@ -293,14 +292,6 @@ def _huggingface(cfg: Config, checks: list[Check]) -> None: break except Exception: pass - try: - if normalize_runtime(cfg.llm_runtime) == "llamacpp": - for e in parse_llamacpp_models(cfg.llamacpp_models_manifest): - if e.url and is_huggingface_url(e.url): - needs_hf = True - break - except Exception: - pass if not cfg.hf_token: checks.append( @@ -309,7 +300,7 @@ def _huggingface(cfg: Config, checks: list[Check]) -> None: True, False, ( - "HF_TOKEN нет — gated GGUF / HF в models.yaml дадут 401. " + "HF_TOKEN нет — gated HF URL в models.yaml дадут 401. " "https://huggingface.co/settings/tokens" if needs_hf else "токена нет (опционально для HF URL / capture fallback)" diff --git a/src/gpu_rent/llm_runtime.py b/src/gpu_rent/llm_runtime.py index 30e0942..c395473 100644 --- a/src/gpu_rent/llm_runtime.py +++ b/src/gpu_rent/llm_runtime.py @@ -1,39 +1,37 @@ -"""Optional LLM runtimes (Ollama / llama.cpp) beside SwarmUI.""" +"""Optional LLM runtime (Ollama) beside SwarmUI.""" from __future__ import annotations from dataclasses import dataclass from pathlib import Path from typing import Any -from urllib.parse import unquote, urlparse import yaml from gpu_rent.paths import ( - llamacpp_models_example_path, - llamacpp_models_manifest_path, ollama_models_example_path, ollama_models_manifest_path, ) -VALID_RUNTIMES = frozenset({"none", "ollama", "llamacpp"}) +VALID_RUNTIMES = frozenset({"none", "ollama"}) OLLAMA_PRESETS: dict[str, list[str]] = { - # Vision + RU/EN + low refusal — best default for swarm-assistent / prompt help with images - "recommended": ["huihui_ai/qwen2.5-vl-abliterated:7b"], - "light": ["huihui_ai/qwen2.5-vl-abliterated:3b"], - "text": ["huihui_ai/qwen2.5-abliterate:7b"], - "stock": ["qwen2.5:7b"], - "alt": ["richardyoung/qwen2.5-7b-instruct-abliterated"], + # Ollama library / community tags only (no GGUF). Tuned for SwarmUI prompt help: + # vision + RU/EN, share VRAM with diffusion on typical 24 GiB. + "recommended": ["huihui_ai/qwen2.5-vl-abliterated:7b"], # ~6 GB, low refusal + "light": ["huihui_ai/qwen2.5-vl-abliterated:3b"], # ~3 GB, tight VRAM + "stock": ["qwen2.5vl:7b"], # official library vision + "text": ["huihui_ai/qwen2.5-abliterate:7b"], # ~5 GB, no vision + "big": ["qwen2.5vl:32b"], # ~21 GB — llm-only or ≥40 GiB free "empty": [], } OLLAMA_PRESET_LABELS: dict[str, str] = { "recommended": "Qwen2.5-VL 7B abliterate (картинки+RU, ~6GB)", - "light": "Qwen2.5-VL 3B abliterate (vision, быстрее, ~3GB)", + "light": "Qwen2.5-VL 3B abliterate (мало VRAM, ~3GB)", + "stock": "официальный qwen2.5vl:7b (library, больше отказов)", "text": "Qwen2.5 7B abliterate text-only (~5GB)", - "stock": "официальный qwen2.5:7b (больше цензуры)", - "alt": "другой text abliterate-пак 7B", + "big": "qwen2.5vl:32b (~21GB; llm-only / большой GPU)", "empty": "только runtime, без pull", "keep": "не менять ollama-models.yaml", } @@ -43,64 +41,9 @@ PRESET_HELP = "\n".join( f"{k} — {v}" for k, v in OLLAMA_PRESET_LABELS.items() if k != "keep" ) -# Each preset entry: {"url": "...gguf", "mmproj": optional vision projector url} -LLAMACPP_PRESETS: dict[str, list[dict[str, str]]] = { - "recommended": [ - { - "url": ( - "https://huggingface.co/mradermacher/Qwen2.5-VL-7B-Instruct-abliterated-GGUF/" - "resolve/main/Qwen2.5-VL-7B-Instruct-abliterated.Q4_K_M.gguf" - ), - "mmproj": ( - "https://huggingface.co/mradermacher/Qwen2.5-VL-7B-Instruct-abliterated-GGUF/" - "resolve/main/Qwen2.5-VL-7B-Instruct-abliterated.mmproj-Q8_0.gguf" - ), - }, - ], - "light": [ - { - "url": ( - "https://huggingface.co/bartowski/Qwen2.5-3B-Instruct-GGUF/" - "resolve/main/Qwen2.5-3B-Instruct-Q4_K_M.gguf" - ), - }, - ], - "text": [ - { - "url": ( - "https://huggingface.co/RichardErkhov/huihui-ai_-_Qwen2.5-7B-Instruct-abliterated-gguf/" - "resolve/main/Qwen2.5-7B-Instruct-abliterated.Q4_K_M.gguf" - ), - }, - ], - "stock": [ - { - "url": ( - "https://huggingface.co/bartowski/Qwen2.5-7B-Instruct-GGUF/" - "resolve/main/Qwen2.5-7B-Instruct-Q4_K_M.gguf" - ), - }, - ], - "empty": [], -} - -LLAMACPP_PRESET_LABELS: dict[str, str] = { - "recommended": "Qwen2.5-VL 7B abliterate + mmproj (~4.7+0.8GB, картинки+RU)", - "light": "Qwen2.5 3B Instruct Q4_K_M text (~2GB)", - "text": "Qwen2.5 7B abliterate text-only Q4_K_M (~4.7GB)", - "stock": "официальный Qwen2.5 7B Instruct Q4_K_M", - "empty": "только llama-server, GGUF вручную", - "keep": "не менять llamacpp-models.yaml", -} - -LLAMACPP_PRESET_HELP = "\n".join( - f"{k} — {v}" for k, v in LLAMACPP_PRESET_LABELS.items() if k != "keep" -) - LLM_RUNTIME_LABELS: dict[str, str] = { "none": "только SwarmUI", "ollama": "Ollama (+ pull моделей)", - "llamacpp": "llama.cpp server (+ GGUF)", } WORKLOAD_LABELS: dict[str, str] = { @@ -113,7 +56,7 @@ WORKLOAD_LABELS: dict[str, str] = { def llm_runtime_menu() -> list: from gpu_rent.prompts import MenuItem - return [MenuItem(k, f"{k} — {LLM_RUNTIME_LABELS[k]}") for k in ("none", "ollama", "llamacpp")] + return [MenuItem(k, f"{k} — {LLM_RUNTIME_LABELS[k]}") for k in ("none", "ollama")] def workload_menu() -> list: @@ -131,53 +74,29 @@ def ollama_preset_menu(*, include_keep: bool = False) -> list: return [MenuItem(k, OLLAMA_PRESET_LABELS.get(k, k)) for k in keys] -def llamacpp_preset_menu(*, include_keep: bool = False) -> list: - from gpu_rent.prompts import MenuItem - - keys = list(LLAMACPP_PRESETS.keys()) - if include_keep: - keys.append("keep") - return [MenuItem(k, LLAMACPP_PRESET_LABELS.get(k, k)) for k in keys] - - @dataclass(frozen=True) class OllamaModelEntry: name: str default: bool = False -@dataclass(frozen=True) -class LlamaCppModelEntry: - url: str - filename: str | None = None - mmproj_url: str | None = None - default: bool = False - - def normalize_runtime(value: str | None) -> str: raw = (value or "none").strip().lower().replace("-", "").replace("_", "") if raw in {"", "none", "off", "no", "0"}: return "none" if raw in {"ollama"}: return "ollama" - if raw in {"llamacpp", "llama", "llamacppserver"}: - return "llamacpp" - raise ValueError(f"неизвестный LLM_RUNTIME={value!r}; жду none|ollama|llamacpp") + raise ValueError(f"неизвестный LLM_RUNTIME={value!r}; жду none|ollama") def decide_runtime( *, flag: str | None, ollama_flag: bool, - llamacpp_flag: bool, from_config: str, ) -> str: - if ollama_flag and llamacpp_flag: - raise ValueError("укажи только --ollama или --llamacpp, не оба") if ollama_flag: return "ollama" - if llamacpp_flag: - return "llamacpp" if flag is not None and str(flag).strip() != "": return normalize_runtime(flag) return normalize_runtime(from_config) @@ -210,65 +129,6 @@ def parse_ollama_models(path: Path) -> list[OllamaModelEntry]: return out -def gguf_filename_from_url(url: str) -> str: - path = unquote(urlparse(url).path) - name = path.rsplit("/", 1)[-1] if path else "" - if name.lower().endswith(".gguf"): - return name - return "model.gguf" - - -# Dead / moved HF mirrors → current resolve URL (same Q4_K_M abliterate weights). -_LLAMACPP_URL_ALIASES: dict[str, str] = { - "https://huggingface.co/bartowski/huihui-ai_Qwen2.5-7B-Instruct-abliterated-GGUF/resolve/main/huihui-ai_Qwen2.5-7B-Instruct-abliterated-Q4_K_M.gguf": ( - "https://huggingface.co/RichardErkhov/huihui-ai_-_Qwen2.5-7B-Instruct-abliterated-gguf/resolve/main/Qwen2.5-7B-Instruct-abliterated.Q4_K_M.gguf" - ), -} - - -def remap_llamacpp_url(url: str) -> str: - """Rewrite known-dead GGUF mirrors so old llamacpp-models.yaml still works.""" - key = (url or "").strip() - return _LLAMACPP_URL_ALIASES.get(key, key) - - -def parse_llamacpp_models(path: Path) -> list[LlamaCppModelEntry]: - if not path.is_file(): - return [] - raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {} - if not isinstance(raw, dict): - return [] - items = raw.get("models") - if items is None: - return [] - if not isinstance(items, list): - raise ValueError(f"{path}: models должен быть списком") - out: list[LlamaCppModelEntry] = [] - for item in items: - if isinstance(item, str): - url = remap_llamacpp_url(item.strip()) - if url: - out.append(LlamaCppModelEntry(url=url)) - continue - if not isinstance(item, dict): - continue - url = remap_llamacpp_url(str(item.get("url") or "").strip()) - if not url: - continue - fname = item.get("filename") - filename = str(fname).strip() if fname else None - mmproj = remap_llamacpp_url(str(item.get("mmproj") or item.get("mmproj_url") or "").strip()) - out.append( - LlamaCppModelEntry( - url=url, - filename=filename or None, - mmproj_url=mmproj or None, - default=bool(item.get("default")), - ) - ) - return out - - def write_ollama_models_preset(path: Path, preset: str) -> None: key = (preset or "recommended").strip().lower() if key not in OLLAMA_PRESETS: @@ -289,32 +149,6 @@ def write_ollama_models_preset(path: Path, preset: str) -> None: path.write_text("\n".join(lines) + "\n", encoding="utf-8") -def write_llamacpp_models_preset(path: Path, preset: str) -> None: - key = (preset or "recommended").strip().lower() - if key not in LLAMACPP_PRESETS: - raise ValueError(f"пресет {preset!r}; варианты: {', '.join(LLAMACPP_PRESETS)}") - specs = LLAMACPP_PRESETS[key] - lines = [ - "# Локальный манифест llama.cpp GGUF (не коммить). Пример: llamacpp-models.example.yaml", - "# url = прямой HTTPS на .gguf; mmproj = projector для vision (Qwen2.5-VL и т.п.).", - "models:", - ] - if not specs: - lines.append(" []") - else: - for i, spec in enumerate(specs): - url = str(spec.get("url") or "").strip() - if not url: - continue - lines.append(f" - url: {url}") - mmproj = str(spec.get("mmproj") or "").strip() - if mmproj: - lines.append(f" mmproj: {mmproj}") - if i == 0: - lines.append(" default: true") - path.write_text("\n".join(lines) + "\n", encoding="utf-8") - - def ensure_ollama_manifest_from_example() -> Path: dest = ollama_models_manifest_path() if dest.is_file(): @@ -327,24 +161,10 @@ def ensure_ollama_manifest_from_example() -> Path: return dest -def ensure_llamacpp_manifest_from_example() -> Path: - dest = llamacpp_models_manifest_path() - if dest.is_file(): - return dest - example = llamacpp_models_example_path() - if example.is_file(): - dest.write_text(example.read_text(encoding="utf-8"), encoding="utf-8") - else: - write_llamacpp_models_preset(dest, "recommended") - return dest - - def llm_local_port(cfg: Any) -> int | None: runtime = normalize_runtime(getattr(cfg, "llm_runtime", "none")) if runtime == "ollama": return int(getattr(cfg, "ollama_local_port", 17811)) - if runtime == "llamacpp": - return int(getattr(cfg, "llamacpp_local_port", 17812)) return None @@ -352,47 +172,9 @@ def llm_remote_port(runtime: str) -> int | None: runtime = normalize_runtime(runtime) if runtime == "ollama": return 11434 - if runtime == "llamacpp": - return 8080 return None -def pick_llamacpp_linux_asset_url(assets: list[dict[str, Any]]) -> str: - """Choose a Linux llama.cpp release asset URL. - - Upstream ships Windows CUDA zips first; never pick win/macos/cudart-only. - Prefer ubuntu+cuda → linux+cuda → ubuntu vulkan x64 → ubuntu x64 CPU. - Mirrored in remote/install_llamacpp.sh (pick_linux_asset_url). - """ - cands: list[tuple[int, str]] = [] - for a in assets: - name = str(a.get("name") or "").lower() - url = str(a.get("browser_download_url") or "") - if not (url.endswith(".zip") or url.endswith(".tar.gz")): - continue - if any(x in name for x in ("win", "macos", "android", "darwin", "xcframework", "-ui.")): - continue - if "cudart" in name: - continue - score = 0 - if "ubuntu" in name and "x64" in name and "cuda" in name: - score = 100 - elif "linux" in name and "cuda" in name: - score = 90 - elif "ubuntu" in name and "vulkan" in name and "x64" in name: - score = 50 - elif "ubuntu" in name and "x64" in name and not any( - x in name for x in ("sycl", "openvino", "arm", "s390", "rocm") - ): - score = 30 - elif "ubuntu" in name or "linux" in name: - score = 10 - if score: - cands.append((score, url)) - cands.sort(key=lambda t: t[0], reverse=True) - return cands[0][1] if cands else "" - - def append_vars_llm_runtime(vars_file: Path, runtime: str) -> None: from gpu_rent.varsfile import upsert_vars diff --git a/src/gpu_rent/manifests.py b/src/gpu_rent/manifests.py index 581a4ab..a7c3025 100644 --- a/src/gpu_rent/manifests.py +++ b/src/gpu_rent/manifests.py @@ -45,7 +45,7 @@ class ModelEntry: url: str | None -VALID_REQUIRES = frozenset({"none", "ollama", "llamacpp", "any-llm"}) +VALID_REQUIRES = frozenset({"none", "ollama", "any-llm"}) @dataclass @@ -107,7 +107,7 @@ def normalize_requires(value: object | None) -> str: if raw in VALID_REQUIRES: return raw raise ConfigError( - f"requires={value!r}: жду none|ollama|llamacpp|any-llm" + f"requires={value!r}: жду none|ollama|any-llm" ) @@ -118,7 +118,7 @@ def repo_matches_runtime(repo: GitRepo, llm_runtime: str) -> bool: if req == "none": return True if req == "any-llm": - return runtime in {"ollama", "llamacpp"} + return runtime == "ollama" return runtime == req diff --git a/src/gpu_rent/paths.py b/src/gpu_rent/paths.py index 3e91575..019aaf0 100644 --- a/src/gpu_rent/paths.py +++ b/src/gpu_rent/paths.py @@ -56,14 +56,6 @@ def ollama_models_example_path() -> Path: return app_root() / "ollama-models.example.yaml" -def llamacpp_models_manifest_path() -> Path: - return app_root() / "llamacpp-models.yaml" - - -def llamacpp_models_example_path() -> Path: - return app_root() / "llamacpp-models.example.yaml" - - def extensions_manifest_path() -> Path: return app_root() / "extensions.yaml" diff --git a/src/gpu_rent/provision.py b/src/gpu_rent/provision.py index a9b2338..c012e80 100644 --- a/src/gpu_rent/provision.py +++ b/src/gpu_rent/provision.py @@ -31,19 +31,6 @@ DATA = "/mnt/swarm_data" # Forwarded to remote install_*.sh (from .env / gpu-rent.vars → os.environ). _OLLAMA_INSTALL_ENV = ("OLLAMA_VERSION", "OLLAMA_SHA256") -_LLAMACPP_INSTALL_ENV = ( - "LLAMACPP_TAG", - "LLAMACPP_ASSET_URL", - "LLAMACPP_SHA256", - "LLAMACPP_BUILD_CUDA", - "LLAMACPP_BACKEND", - "LLAMACPP_FORCE_REINSTALL", - "LLAMACPP_NGL", - "LLAMACPP_CTX", - "LLAMACPP_HOST", - "LLAMACPP_PORT", - "LLAMACPP_EXTRA_ARGS", -) def _remote_llm_env(cfg: Config, *keys: str) -> dict[str, str]: @@ -450,7 +437,7 @@ def provision_llm(cfg: Config, host: str, log: Log) -> None: check=False, ) - # Always drop the other runtime so VRAM is not held by a leftover unit. + # Drop LLM units that should not hold VRAM for this runtime. if runtime == "none": log("LLM: none — останавливаю gpu-rent-ollama / gpu-rent-llamacpp если были") _stop_units("gpu-rent-ollama", "gpu-rent-llamacpp") @@ -490,71 +477,8 @@ def provision_llm(cfg: Config, host: str, log: Log) -> None: timeout=7200, log=log, ) - elif runtime == "llamacpp": - _stop_units("gpu-rent-ollama") - from gpu_rent.llm_runtime import ( - gguf_filename_from_url, - parse_llamacpp_models, - ) - - entries = parse_llamacpp_models(cfg.llamacpp_models_manifest) - defaults = [e for e in entries if e.default] - if defaults: - log( - "llama.cpp preferred: " - f"{defaults[0].filename or gguf_filename_from_url(defaults[0].url)}" - ) - if entries: - jobs = [] - for e in entries: - jobs.append( - { - "url": e.url, - "filename": e.filename or gguf_filename_from_url(e.url), - } - ) - if e.mmproj_url: - jobs.append( - { - "url": e.mmproj_url, - "filename": gguf_filename_from_url(e.mmproj_url), - } - ) - put_text( - cfg, host, "/tmp/gpu-rent-llamacpp-models.json", json.dumps(jobs, indent=2) - ) - hf = (cfg.hf_token or "").strip() - if hf: - put_text(cfg, host, "/tmp/gpu-rent-hf.token", hf + "\n", mode=0o600) - else: - log( - "⚠ HF_TOKEN не задан — gated GGUF (abliterated и др.) часто дают 401. " - "Добавь HF_TOKEN=hf_… в .env → https://huggingface.co/settings/tokens" - ) - log(f"llama.cpp: скачиваю {len(jobs)} GGUF из манифеста") - run_python( - cfg, - host, - _pkg_text("llamacpp_fetch.py"), - remote_path="/tmp/gpu-rent-llamacpp_fetch.py", - timeout=7200, - log=log, - ) - else: - log("llamacpp-models.yaml пуст — GGUF skip (положи вручную)") - log("LLM: ставим/запускаем llama.cpp server") - import os - - # Vulkan finishes in seconds; CUDA compile needs up to ~15–20 min. - run_script_sudo( - cfg, - host, - _pkg_text("install_llamacpp.sh"), - remote_path="/tmp/gpu-rent-install_llamacpp.sh", - timeout=3600, - env=_remote_llm_env(cfg, *_LLAMACPP_INSTALL_ENV), - log=log, - ) + else: + raise CloudError(f"неизвестный LLM_RUNTIME={runtime!r}") st = load_state() st.notes = dict(st.notes or {}) st.notes["llm_runtime"] = runtime @@ -614,7 +538,7 @@ def provision_vm( rt = normalize_runtime(cfg.llm_runtime) if not swarm and rt == "none": raise CloudError( - "llm-only: нужен LLM_RUNTIME=ollama|llamacpp (или --ollama / --llamacpp)" + "llm-only: нужен LLM_RUNTIME=ollama (или --ollama / --llm ollama)" ) # Arm ASAP so mid-provision failures still leave auto-stop on the VM. @@ -692,6 +616,4 @@ def provision_vm( log("SwarmUI слушает 127.0.0.1:7801 — gpu-rent tunnel") if rt == "ollama": log(f"Ollama API → localhost:{cfg.ollama_local_port} (туннель)") - elif rt == "llamacpp": - log(f"llama.cpp → localhost:{cfg.llamacpp_local_port} (туннель)") log("Hold killer: gpu-rent hold | Стоп GPU: gpu-rent stop") diff --git a/src/gpu_rent/ready.py b/src/gpu_rent/ready.py index 540eb3a..bf61636 100644 --- a/src/gpu_rent/ready.py +++ b/src/gpu_rent/ready.py @@ -95,7 +95,6 @@ def unit_active(name): checks = [] want_swarm = WANT_SWARM want_ollama = WANT_OLLAMA -want_llama = WANT_LLAMA if want_swarm: ok, detail = http_ok("http://127.0.0.1:7801/") @@ -131,18 +130,6 @@ if want_ollama: "unit": unit_active("gpu-rent-ollama"), }) -if want_llama: - ok, detail = http_ok("http://127.0.0.1:8080/health") - if not ok: - ok2, d2 = http_ok("http://127.0.0.1:8080/v1/models") - ok, detail = ok2, d2 - checks.append({ - "name": "llamacpp", - "ok": ok, - "detail": detail, - "unit": unit_active("gpu-rent-llamacpp"), - }) - print(json.dumps({"checks": checks}, ensure_ascii=False)) ''' @@ -192,18 +179,17 @@ def wait_backend_idle( ) -def _expected_services(cfg: Config) -> tuple[bool, bool, bool]: +def _expected_services(cfg: Config) -> tuple[bool, bool]: swarm = bool(getattr(cfg, "enable_swarmui", True)) rt = normalize_runtime(getattr(cfg, "llm_runtime", "none")) - return swarm, rt == "ollama", rt == "llamacpp" + return swarm, rt == "ollama" def _probe_vm_once(cfg: Config, host: str) -> list[ServiceCheck]: - want_swarm, want_ollama, want_llama = _expected_services(cfg) + want_swarm, want_ollama = _expected_services(cfg) script = ( _REMOTE_STACK_PROBE.replace("WANT_SWARM", "True" if want_swarm else "False") .replace("WANT_OLLAMA", "True" if want_ollama else "False") - .replace("WANT_LLAMA", "True" if want_llama else "False") ) out = run_ssh( cfg, @@ -251,8 +237,8 @@ def verify_stack_on_vm( raise_on_fail: bool = True, ) -> list[ServiceCheck]: """Poll until every enabled service answers on the VM loopback.""" - want_swarm, want_ollama, want_llama = _expected_services(cfg) - if not (want_swarm or want_ollama or want_llama): + want_swarm, want_ollama = _expected_services(cfg) + if not (want_swarm or want_ollama): log("проверка стека: нечего ждать (swarm off, LLM none)") return [] @@ -261,8 +247,6 @@ def verify_stack_on_vm( names.append("SwarmUI :7801") if want_ollama: names.append("Ollama :11434") - if want_llama: - names.append("llama.cpp :8080") log(f"проверка на VM: {', '.join(names)}") deadline = time.time() + timeout @@ -447,7 +431,7 @@ def verify_stack_local( raise_on_fail: bool = True, ) -> list[ServiceCheck]: """After tunnel: local ports + light HTTP for enabled services.""" - want_swarm, want_ollama, want_llama = _expected_services(cfg) + want_swarm, want_ollama = _expected_services(cfg) targets: list[tuple[str, int, str | None]] = [] if want_swarm: targets.append(("swarmui", int(cfg.swarmui_local_port), None)) @@ -459,9 +443,6 @@ def verify_stack_local( f"http://127.0.0.1:{cfg.ollama_local_port}/api/tags", ) ) - if want_llama: - p = int(cfg.llamacpp_local_port) - targets.append(("llamacpp", p, f"http://127.0.0.1:{p}/health")) if not targets: return [] @@ -481,8 +462,6 @@ def verify_stack_local( continue if url: ok, detail = _http_local(url) - if not ok and name == "llamacpp": - ok, detail = _http_local(f"http://127.0.0.1:{port}/v1/models") last.append(ServiceCheck(name, ok, detail, "local")) else: ok, detail = _http_local(f"http://127.0.0.1:{port}/") diff --git a/src/gpu_rent/remote/bootstrap.sh b/src/gpu_rent/remote/bootstrap.sh index 92c0b7c..5321185 100644 --- a/src/gpu_rent/remote/bootstrap.sh +++ b/src/gpu_rent/remote/bootstrap.sh @@ -108,8 +108,7 @@ mkdir -p \ "${DATA_ROOT}/Extensions" \ "${DATA_ROOT}/DLNodes" \ "${DATA_ROOT}/CustomWorkflows" \ - "${DATA_ROOT}/ollama" \ - "${DATA_ROOT}/llamacpp/models" + "${DATA_ROOT}/ollama" # LLM-only: data disk + tools, no SwarmUI clone / unit. if [[ "${GPU_RENT_SKIP_SWARMUI:-0}" == "1" ]]; then diff --git a/src/gpu_rent/remote/idle_killer.py b/src/gpu_rent/remote/idle_killer.py index c31fad6..ff27366 100644 --- a/src/gpu_rent/remote/idle_killer.py +++ b/src/gpu_rent/remote/idle_killer.py @@ -107,7 +107,7 @@ def swarm_busy(swarm_url: str, timeout: float = 8.0) -> tuple[bool, str]: def llm_busy(timeout: float = 3.0) -> tuple[bool, str]: - """Ollama pull / loaded models or llama.cpp with a model count as busy.""" + """Ollama pull / loaded models count as busy.""" pull_marker = DATA / ".gpu-rent-ollama-pulling" if pull_marker.is_file(): try: @@ -138,24 +138,6 @@ def llm_busy(timeout: float = 3.0) -> tuple[bool, str]: return True, f"ollama running {names}" except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError, json.JSONDecodeError, OSError): pass - # llama.cpp: slots in use - try: - req = urllib.request.Request("http://127.0.0.1:8080/health", method="GET") - with urllib.request.urlopen(req, timeout=timeout, context=ctx) as resp: - if getattr(resp, "status", 200) == 200: - try: - req2 = urllib.request.Request("http://127.0.0.1:8080/props", method="GET") - with urllib.request.urlopen(req2, timeout=timeout, context=ctx) as resp2: - props = json.loads(resp2.read().decode("utf-8")) - total = int(props.get("total_slots") or 0) - avail = int(props.get("available_slots") or total) - in_use = total - avail if total else 0 - if in_use > 0: - return True, f"llamacpp slots_in_use={in_use}" - except Exception: - pass - except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError, OSError): - pass return False, "llm idle" diff --git a/src/gpu_rent/remote/install_llamacpp.sh b/src/gpu_rent/remote/install_llamacpp.sh deleted file mode 100644 index ab82171..0000000 --- a/src/gpu_rent/remote/install_llamacpp.sh +++ /dev/null @@ -1,471 +0,0 @@ -#!/usr/bin/env bash -# Install llama-server for OpenAI-compatible API on loopback :8080. -# -# Default: official Linux release asset (Ubuntu Vulkan — GPU without compile). -# CUDA source build only as last resort (or LLAMACPP_BUILD_CUDA=1). -# Pin: LLAMACPP_TAG=b10545 LLAMACPP_ASSET_URL=... LLAMACPP_SHA256=... -set -euo pipefail - -SWARM_USER="${SWARM_USER:-ubuntu}" -DATA_ROOT="/mnt/swarm_data" -LLAMA_ROOT="${DATA_ROOT}/llamacpp" -MODELS_DIR="${LLAMA_ROOT}/models" -BIN_DIR="${LLAMA_ROOT}/bin" -SRC_DIR="${LLAMA_ROOT}/src" -STAMP="${BIN_DIR}/.build-id" -UNIT="gpu-rent-llamacpp" -REPO="https://github.com/ggml-org/llama.cpp.git" -API_BASE="https://api.github.com/repos/ggml-org/llama.cpp" - -log() { echo "[gpu-rent-llamacpp] $*" >&2; } - -if [[ "$(id -u)" -ne 0 ]]; then - echo "нужен root" >&2 - exit 1 -fi - -mkdir -p "$MODELS_DIR" "$BIN_DIR" -chown -R "${SWARM_USER}:${SWARM_USER}" "$LLAMA_ROOT" - -SERVER_BIN="${BIN_DIR}/llama-server" -LLAMACPP_TAG="${LLAMACPP_TAG:-}" -LLAMACPP_ASSET_URL="${LLAMACPP_ASSET_URL:-}" -LLAMACPP_SHA256="${LLAMACPP_SHA256:-}" -# 1 = force CUDA compile; 0 = never compile (Vulkan/CPU prebuilt only) -LLAMACPP_BUILD_CUDA="${LLAMACPP_BUILD_CUDA:-}" -# auto | cuda | vulkan — default auto: CUDA if nvcc already on VM, else Vulkan prebuilt -LLAMACPP_BACKEND="${LLAMACPP_BACKEND:-auto}" -LLAMACPP_FORCE_REINSTALL="${LLAMACPP_FORCE_REINSTALL:-}" -LLAMACPP_NGL="${LLAMACPP_NGL:-}" -LLAMACPP_CTX="${LLAMACPP_CTX:-}" -LLAMACPP_HOST="${LLAMACPP_HOST:-127.0.0.1}" -LLAMACPP_PORT="${LLAMACPP_PORT:-8080}" -LLAMACPP_EXTRA_ARGS="${LLAMACPP_EXTRA_ARGS:-}" - -have_nvcc() { - if command -v nvcc >/dev/null 2>&1; then - return 0 - fi - if [[ -x /usr/local/cuda/bin/nvcc ]]; then - export PATH="/usr/local/cuda/bin:${PATH}" - return 0 - fi - return 1 -} - -# Prefer CUDA when toolkit already present (GPU images / prior up). Vulkan = fast no-compile. -want_cuda_build() { - case "${LLAMACPP_BUILD_CUDA}" in - 1|yes|true) return 0 ;; - 0|no|false) return 1 ;; - esac - case "${LLAMACPP_BACKEND}" in - cuda) return 0 ;; - vulkan) return 1 ;; - *) - if have_nvcc; then - return 0 - fi - return 1 - ;; - esac -} - -if [[ "${LLAMACPP_FORCE_REINSTALL}" == "1" ]]; then - log "LLAMACPP_FORCE_REINSTALL=1 — удаляю старый бинарь" - rm -f "$SERVER_BIN" "$STAMP" -fi - -# Upgrade path: previous default was Vulkan prebuilt; if nvcc is here, prefer CUDA. -if [[ -x "$SERVER_BIN" && -f "$STAMP" && "${LLAMACPP_BACKEND}" != "vulkan" && "${LLAMACPP_BUILD_CUDA}" != "0" ]]; then - if grep -q '^asset:' "$STAMP" 2>/dev/null && want_cuda_build; then - log "был Vulkan/CPU prebuilt, nvcc есть — пересобираю CUDA (лучше на NVIDIA)" - rm -f "$SERVER_BIN" "$STAMP" - fi -fi - -# Prefer ubuntu CUDA (rare) → vulkan → cpu. Never Windows/macOS/cudart-only. -pick_linux_asset_url() { - python3 -c ' -import json,sys -data=json.load(sys.stdin) -assets=data.get("assets") or [] -cands=[] -for a in assets: - n=(a.get("name") or "").lower() - u=a.get("browser_download_url") or "" - if not (u.endswith(".zip") or u.endswith(".tar.gz")): - continue - if any(x in n for x in ("win","macos","android","darwin","xcframework","-ui.")): - continue - if "cudart" in n: - continue - score=0 - if "ubuntu" in n and "x64" in n and "cuda" in n: - score=100 - elif "linux" in n and "cuda" in n: - score=90 - elif "ubuntu" in n and "vulkan" in n and "x64" in n: - score=50 - elif "ubuntu" in n and "x64" in n and not any( - x in n for x in ("sycl","openvino","arm","s390","rocm") - ): - score=30 - elif "ubuntu" in n or "linux" in n: - score=10 - if score: - cands.append((score, u, n)) -cands.sort(reverse=True) -print(cands[0][1] if cands else "") -' -} - -resolve_release_tag() { - # stdout = tag only (no log lines — callers capture via $()) - if [[ -n "$LLAMACPP_TAG" ]]; then - echo "$LLAMACPP_TAG" - return - fi - curl -fsSL "${API_BASE}/releases/latest" | python3 -c \ - 'import json,sys; print(json.load(sys.stdin).get("tag_name") or "")' -} - -cuda_architectures() { - python3 - <<'PY' -import json -from pathlib import Path -p = Path("/mnt/swarm_data/.gpu-rent-gpu.json") -cap = "8.9" -if p.is_file(): - try: - cap = str(json.loads(p.read_text()).get("compute_cap") or cap) - except Exception: - pass -parts = cap.split(".") -try: - maj, mnr = int(parts[0]), int(parts[1]) if len(parts) > 1 else 0 - print(f"{maj}{mnr}") -except ValueError: - print("89") -PY -} - -ensure_build_deps() { - export DEBIAN_FRONTEND=noninteractive - apt-get install -y -qq \ - cmake build-essential git curl ca-certificates \ - libcurl4-openssl-dev >/dev/null - if command -v nvcc >/dev/null 2>&1; then - return 0 - fi - if [[ -x /usr/local/cuda/bin/nvcc ]]; then - export PATH="/usr/local/cuda/bin:${PATH}" - return 0 - fi - log "ставлю nvidia-cuda-toolkit (нужен nvcc)…" - apt-get install -y -qq nvidia-cuda-toolkit >/dev/null - if command -v nvcc >/dev/null 2>&1; then - return 0 - fi - if [[ -x /usr/local/cuda/bin/nvcc ]]; then - export PATH="/usr/local/cuda/bin:${PATH}" - return 0 - fi - return 1 -} - -ensure_vulkan_runtime() { - if ldconfig -p 2>/dev/null | grep -q 'libvulkan\.so'; then - return 0 - fi - export DEBIAN_FRONTEND=noninteractive - log "ставлю libvulkan1 (для Ubuntu Vulkan prebuilt)…" - apt-get install -y -qq libvulkan1 mesa-vulkan-drivers >/dev/null 2>&1 || \ - apt-get install -y -qq libvulkan1 >/dev/null 2>&1 || true -} - -install_from_archive_url() { - local url="$1" - local tmp kind - tmp="$(mktemp -d)" - ( - cd "$tmp" - log "скачиваю prebuilt: $url" - curl -fL --progress-bar "$url" -o pkg.bin - if [[ -n "$LLAMACPP_SHA256" ]]; then - echo "${LLAMACPP_SHA256} pkg.bin" | sha256sum -c - - else - log "WARN: LLAMACPP_SHA256 не задан — checksum skip" - fi - mkdir -p out - # Do NOT grep -i zip — that matches "gzip" and breaks .tar.gz. - kind="$(file -b pkg.bin 2>/dev/null || true)" - case "$url" in - *.zip) - apt-get install -y -qq unzip >/dev/null 2>&1 || true - unzip -qo pkg.bin -d out - ;; - *) - if [[ "$kind" == Zip\ archive* ]] || [[ "$kind" == *"Zip archive"* ]]; then - apt-get install -y -qq unzip >/dev/null 2>&1 || true - unzip -qo pkg.bin -d out - else - tar -xaf pkg.bin -C out 2>/dev/null \ - || tar -xzf pkg.bin -C out 2>/dev/null \ - || tar -xf pkg.bin -C out - fi - ;; - esac - local found - found="$(find out -type f -name 'llama-server' | head -n1 || true)" - if [[ -z "$found" ]]; then - found="$(find out -type f -name 'server' | head -n1 || true)" - fi - if [[ -z "$found" ]]; then - log "в архиве нет llama-server (file says: ${kind:-unknown})" - exit 1 - fi - install -m 755 "$found" "$SERVER_BIN" - # Shared libs next to binary (release tarballs ship .so alongside). - find out -type f \( -name '*.so' -o -name '*.so.*' \) -print0 2>/dev/null \ - | while IFS= read -r -d '' so; do - install -m 755 "$so" "${BIN_DIR}/$(basename "$so")" - done - chown -R "${SWARM_USER}:${SWARM_USER}" "$BIN_DIR" - ) - local rc=$? - rm -rf "$tmp" - return "$rc" -} - -install_linux_release() { - local tag="$1" - local api url - api="${API_BASE}/releases/tags/${tag}" - url="$(curl -fsSL "$api" | pick_linux_asset_url)" - if [[ -z "$url" ]]; then - log "в release ${tag} нет Linux-ассета" - return 1 - fi - if [[ "$url" == *vulkan* ]]; then - log "беру Ubuntu Vulkan prebuilt (GPU без compile; CUDA-сборка — LLAMACPP_BUILD_CUDA=1)" - ensure_vulkan_runtime - elif [[ "$url" == *cuda* ]]; then - log "беру Linux CUDA prebuilt" - else - log "WARN: Linux prebuilt без GPU backend (CPU) — ${url##*/}" - fi - if ! install_from_archive_url "$url"; then - return 1 - fi - echo "asset:${tag}" >"$STAMP" - chown "${SWARM_USER}:${SWARM_USER}" "$STAMP" -} - -# Quiet CUDA build: no cmake spam; heartbeat every 30s with last %. -build_cuda_from_source() { - local tag="$1" - local arch build_log pid pct line - arch="$(cuda_architectures)" - build_log="${LLAMA_ROOT}/build-cuda.log" - log "крайний случай: сборка CUDA из исходников (tag=${tag}, arch=${arch}, 5–15 мин)…" - log "полный лог: ${build_log}" - if ! ensure_build_deps; then - log "нет nvcc — CUDA-сборку пропускаем" - return 1 - fi - log "nvcc $(nvcc --version 2>/dev/null | tail -n1 || echo '?')" - - mkdir -p "$SRC_DIR" - export GIT_TERMINAL_PROMPT=0 - if [[ -d "${SRC_DIR}/.git" ]]; then - git -C "$SRC_DIR" -c advice.detachedHead=false fetch --depth 1 origin tag "$tag" 2>>"$build_log" || true - if ! git -C "$SRC_DIR" -c advice.detachedHead=false checkout -f "$tag" >>"$build_log" 2>&1; then - rm -rf "$SRC_DIR" - git -c advice.detachedHead=false clone --depth 1 --branch "$tag" "$REPO" "$SRC_DIR" >>"$build_log" 2>&1 - fi - else - rm -rf "$SRC_DIR" - git -c advice.detachedHead=false clone --depth 1 --branch "$tag" "$REPO" "$SRC_DIR" >>"$build_log" 2>&1 - fi - - cmake -S "$SRC_DIR" -B "${SRC_DIR}/build" \ - -DCMAKE_BUILD_TYPE=Release \ - -DGGML_CUDA=ON \ - -DCMAKE_CUDA_ARCHITECTURES="${arch}" \ - -DLLAMA_BUILD_SERVER=ON \ - -DLLAMA_BUILD_UI=OFF \ - -DLLAMA_USE_PREBUILT_UI=OFF \ - -DGGML_CCACHE=OFF \ - >>"$build_log" 2>&1 - - # Background build + heartbeat (keeps SSH stream alive without 200 cmake lines). - cmake --build "${SRC_DIR}/build" -j"$(nproc)" --target llama-server \ - >>"$build_log" 2>&1 & - pid=$! - while kill -0 "$pid" 2>/dev/null; do - pct="$(grep -oE '\[[[:space:]]*[0-9]+%\]' "$build_log" 2>/dev/null | tail -n1 || true)" - line="$(grep -E 'Building CUDA|Built target|Linking' "$build_log" 2>/dev/null | tail -n1 || true)" - if [[ -n "$pct" ]]; then - log "сборка CUDA ещё идёт… ${pct}${line:+ · ${line}}" - else - log "сборка CUDA ещё идёт… (cmake/nvcc, см. build.log)" - fi - sleep 30 - done - if ! wait "$pid"; then - log "сборка упала — хвост ${build_log}:" - tail -n 40 "$build_log" >&2 || true - return 1 - fi - - local built="${SRC_DIR}/build/bin/llama-server" - if [[ ! -x "$built" ]]; then - log "сборка не дала ${built}" - return 1 - fi - install -m 755 "$built" "$SERVER_BIN" - # CUDA build may need libs from build/bin - find "${SRC_DIR}/build/bin" -maxdepth 1 -type f \( -name '*.so' -o -name '*.so.*' \) -print0 2>/dev/null \ - | while IFS= read -r -d '' so; do - install -m 755 "$so" "${BIN_DIR}/$(basename "$so")" - done - chown -R "${SWARM_USER}:${SWARM_USER}" "$BIN_DIR" - echo "cuda:${tag}:${arch}" >"$STAMP" - chown "${SWARM_USER}:${SWARM_USER}" "$STAMP" - log "CUDA binary → ${SERVER_BIN}" - return 0 -} - -normalize_tag() { - printf '%s' "$1" | tr -d '\r' | head -n1 | awk 'NF{print; exit}' -} - -if [[ -x "$SERVER_BIN" ]]; then - log "llama-server уже есть: ${SERVER_BIN}" -else - if [[ -n "$LLAMACPP_ASSET_URL" ]]; then - log "скачиваю по LLAMACPP_ASSET_URL…" - install_from_archive_url "$LLAMACPP_ASSET_URL" - echo "asset-url" >"$STAMP" - chown "${SWARM_USER}:${SWARM_USER}" "$STAMP" - else - if [[ -z "$LLAMACPP_TAG" ]]; then - log "WARN: LLAMACPP_TAG не задан — latest (см. docs/llm.md)" - fi - tag="$(normalize_tag "$(resolve_release_tag)")" - if [[ -z "$tag" || "$tag" == *" "* || "$tag" == *"["* ]]; then - log "не удалось определить release tag (got: ${tag:-empty})" - exit 1 - fi - - installed=0 - if want_cuda_build; then - log "backend: CUDA (nvcc есть или LLAMACPP_BACKEND/BUILD_CUDA) — сборка, Vulkan только если упадёт" - if build_cuda_from_source "$tag"; then - installed=1 - else - log "CUDA-сборка не вышла — fallback на Linux prebuilt (Vulkan/CPU)" - if install_linux_release "$tag"; then - installed=1 - fi - fi - else - log "backend: Linux prebuilt (нет nvcc / LLAMACPP_BACKEND=vulkan) — без compile" - if install_linux_release "$tag"; then - installed=1 - else - log "prebuilt не вышел — крайний случай: CUDA из исходников" - if build_cuda_from_source "$tag"; then - installed=1 - fi - fi - fi - if [[ "$installed" != "1" ]]; then - log "не удалось поставить llama-server" - exit 1 - fi - fi -fi - -if [[ ! -x "$SERVER_BIN" ]]; then - log "нет исполняемого ${SERVER_BIN}" - exit 1 -fi - -# Prefer a weights GGUF (skip mmproj), then attach --mmproj if present. -MODEL_ARG="" -MMPROJ_ARG="" -FIRST_GGUF="$( - find "$MODELS_DIR" -type f \( -name '*.gguf' -o -name '*.GGUF' \) \ - ! -iname '*mmproj*' 2>/dev/null | head -n1 || true -)" -MMPROJ_GGUF="$( - find "$MODELS_DIR" -type f \( -iname '*mmproj*.gguf' -o -iname '*mmproj*.GGUF' \) \ - 2>/dev/null | head -n1 || true -)" -if [[ -n "$FIRST_GGUF" ]]; then - MODEL_ARG="-m ${FIRST_GGUF}" - log "модель ${FIRST_GGUF}" -else - log "нет GGUF в ${MODELS_DIR} — положи файл вручную и systemctl restart ${UNIT}" -fi -if [[ -n "$MMPROJ_GGUF" ]]; then - MMPROJ_ARG="--mmproj ${MMPROJ_GGUF}" - log "mmproj ${MMPROJ_GGUF}" -fi - -# GPU layers: share card with Swarm — full offload on mid+, leave headroom on low. -NGL=99 -CTX=8192 -if [[ -f "${DATA_ROOT}/.gpu-rent-gpu.json" ]]; then - eval "$(python3 - <<'PY' -import json -from pathlib import Path -gpu=json.loads(Path("/mnt/swarm_data/.gpu-rent-gpu.json").read_text()) -vram=int(gpu.get("vram_mib") or 0) -gib=vram/1024.0 -if gib < 16: - print("NGL=40"); print("CTX=4096") -elif gib < 24: - print("NGL=99"); print("CTX=8192") -elif gib < 48: - print("NGL=99"); print("CTX=16384") -else: - print("NGL=99"); print("CTX=32768") -PY -)" || true -fi -# Explicit overrides from gpu-rent.vars / .env (forwarded by provision). -if [[ -n "$LLAMACPP_NGL" ]]; then - NGL="$LLAMACPP_NGL" -fi -if [[ -n "$LLAMACPP_CTX" ]]; then - CTX="$LLAMACPP_CTX" -fi -log "llama.cpp -ngl ${NGL} -c ${CTX} host=${LLAMACPP_HOST} port=${LLAMACPP_PORT}${LLAMACPP_EXTRA_ARGS:+ extra=${LLAMACPP_EXTRA_ARGS}}" - -cat >/etc/systemd/system/${UNIT}.service < str: - n = float(n) - for unit, div in (("GB", 1024**3), ("MB", 1024**2), ("KB", 1024), ("B", 1)): - if n >= div or unit == "B": - if unit == "B": - return f"{int(n)}B" - return f"{n / div:.1f}{unit}" - return f"{n:.0f}B" - - -def progress_line( - label: str, - done: int, - total: int | None, - speed: float, - *, - width: int = 22, -) -> str: - if total and total > 0: - pct = min(100.0, 100.0 * done / total) - filled = int(width * done / total) - filled = min(width, max(0, filled)) - bar = "#" * filled + "-" * (width - filled) - return ( - f"{label} [{bar}] {pct:5.1f}% " - f"{fmt_bytes(done)}/{fmt_bytes(total)} {fmt_bytes(speed)}/s" - ) - return f"{label} {fmt_bytes(done)} {fmt_bytes(speed)}/s" - - -class DownloadProgress: - def __init__(self, label: str, total: int | None) -> None: - self.label = label - self.total = total if total and total > 0 else None - self.done = 0 - self.t0 = time.monotonic() - self.last_print = 0.0 - - def add(self, n: int) -> None: - self.done += n - now = time.monotonic() - if now - self.last_print < 1.0 and not ( - self.total is not None and self.done >= self.total - ): - return - self.last_print = now - self._emit() - - def finish(self) -> None: - self._emit(final=True) - - def _emit(self, *, final: bool = False) -> None: - elapsed = max(time.monotonic() - self.t0, 0.001) - line = progress_line(self.label, self.done, self.total, self.done / elapsed) - if final: - print(line, flush=True) - else: - print(line, end="\r", flush=True) - - -def download(url: str, dest: Path, headers: dict[str, str], *, label: str) -> None: - partial = dest.with_suffix(dest.suffix + ".partial") - - class StripAuthRedirect(urllib.request.HTTPRedirectHandler): - def redirect_request(self, req, fp, code, msg, headers_resp, newurl): - new = urllib.request.HTTPRedirectHandler.redirect_request( - self, req, fp, code, msg, headers_resp, newurl - ) - if new is None: - return None - host = (urllib.parse.urlparse(new.full_url).hostname or "").lower() - # Hub needs Bearer; CDN (cdn-lfs.*) is pre-signed — drop Authorization. - if host in {"huggingface.co", "hf.co"}: - return new - return urllib.request.Request( - new.full_url, headers={"User-Agent": headers.get("User-Agent", "gpu-rent/1")} - ) - - opener = urllib.request.build_opener(StripAuthRedirect) - req = urllib.request.Request(url, headers=headers) - with opener.open(req, timeout=600) as resp, partial.open("wb") as out: - cl = resp.headers.get("Content-Length") - try: - total_n = int(cl) if cl else None - except ValueError: - total_n = None - prog = DownloadProgress(label, total_n) - while True: - chunk = resp.read(1024 * 1024) - if not chunk: - break - out.write(chunk) - prog.add(len(chunk)) - prog.finish() - partial.replace(dest) - - -def main() -> int: - if TOKEN_FILE.is_file(): - try: - os.environ["HF_TOKEN"] = TOKEN_FILE.read_text(encoding="utf-8").strip() - finally: - try: - TOKEN_FILE.unlink(missing_ok=True) - except OSError: - pass - if not JOBS.is_file(): - print("no jobs file") - return 1 - jobs = json.loads(JOBS.read_text(encoding="utf-8")) - if not isinstance(jobs, list) or not jobs: - print("llamacpp fetch: пустой список — skip") - return 0 - MODELS_DIR.mkdir(parents=True, exist_ok=True) - token = (os.environ.get("HF_TOKEN") or os.environ.get("HUGGING_FACE_HUB_TOKEN") or "").strip() - failed = 0 - for i, job in enumerate(jobs, 1): - if not isinstance(job, dict): - continue - url = str(job.get("url") or "").strip() - name = str(job.get("filename") or "").strip() - if not url: - continue - if not name: - name = url.rstrip("/").rsplit("/", 1)[-1] or "model.gguf" - dest = MODELS_DIR / name - prefix = f"[{i}/{len(jobs)}]" - if dest.is_file() and dest.stat().st_size > 1_000_000: - print(f"{prefix} уже есть {name} ({fmt_bytes(dest.stat().st_size)})") - continue - print(f"{prefix} качаю {name}", flush=True) - headers = {"User-Agent": "gpu-rent/1"} - if token: - headers["Authorization"] = f"Bearer {token}" - try: - download(url, dest, headers, label=f"{prefix} {name}") - print(f"{prefix} ok {name} ({fmt_bytes(dest.stat().st_size)})") - except (urllib.error.URLError, urllib.error.HTTPError, OSError, TimeoutError) as exc: - failed += 1 - msg = str(exc) - if "401" in msg or "403" in msg: - if not token: - msg += ( - " — нет HF_TOKEN: добавь в .env " - "(https://huggingface.co/settings/tokens) и прими условия репо" - ) - else: - msg += ( - " — токен есть, но отказано: проверь scopes / " - "Accept license на странице модели" - ) - print(f"FAIL {name}: {msg}") - try: - dest.with_suffix(dest.suffix + ".partial").unlink(missing_ok=True) - except OSError: - pass - if failed: - return 1 - print("llamacpp fetch ok") - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/src/gpu_rent/setup_wizard.py b/src/gpu_rent/setup_wizard.py index d36aeb4..6cf68c0 100644 --- a/src/gpu_rent/setup_wizard.py +++ b/src/gpu_rent/setup_wizard.py @@ -8,21 +8,16 @@ from pathlib import Path from gpu_rent.llm_runtime import ( append_vars_llm_runtime, - ensure_llamacpp_manifest_from_example, ensure_ollama_manifest_from_example, - llamacpp_preset_menu, llm_runtime_menu, normalize_runtime, ollama_preset_menu, - write_llamacpp_models_preset, write_ollama_models_preset, ) from gpu_rent.paths import ( app_root, env_path, extensions_manifest_path, - llamacpp_models_example_path, - llamacpp_models_manifest_path, models_manifest_path, ollama_models_example_path, ollama_models_manifest_path, @@ -66,12 +61,6 @@ def run_setup( _copy_if_missing( ollama_models_example_path(), ollama_models_manifest_path(), "ollama-models.yaml", log ) - _copy_if_missing( - llamacpp_models_example_path(), - llamacpp_models_manifest_path(), - "llamacpp-models.yaml", - log, - ) runtime = llm if runtime is None: @@ -107,24 +96,6 @@ def run_setup( else: write_ollama_models_preset(ollama_models_manifest_path(), preset) log(f"ollama-models.yaml пресет={preset}") - elif runtime == "llamacpp": - preset = ollama_preset - if preset is None and ask: - preset = prompt_menu( - "llama.cpp GGUF", - llamacpp_preset_menu(include_keep=False), - default="recommended", - ask=ask, - show=log, - ) - if preset is None: - preset = "recommended" - if preset.strip().lower() in {"keep", "example", ""}: - ensure_llamacpp_manifest_from_example() - log("llamacpp-models.yaml из example") - else: - write_llamacpp_models_preset(llamacpp_models_manifest_path(), preset) - log(f"llamacpp-models.yaml пресет={preset}") do_wd = install_watchdog if do_wd is None and confirm: diff --git a/src/gpu_rent/tunnel.py b/src/gpu_rent/tunnel.py index 0b17342..7c16393 100644 --- a/src/gpu_rent/tunnel.py +++ b/src/gpu_rent/tunnel.py @@ -79,8 +79,6 @@ def tunnel_forwards(cfg: Config) -> list[tuple[int, int]]: runtime = normalize_runtime(cfg.llm_runtime) if runtime == "ollama": pairs.append((cfg.ollama_local_port, 11434)) - elif runtime == "llamacpp": - pairs.append((cfg.llamacpp_local_port, 8080)) if not pairs: # Failsafe: at least SwarmUI port so tunnel isn't empty. pairs.append((cfg.swarmui_local_port, 7801)) @@ -189,8 +187,6 @@ def run_tunnel( open_url = f"http://127.0.0.1:{cfg.swarmui_local_port}" elif runtime == "ollama": open_url = f"http://127.0.0.1:{cfg.ollama_local_port}" - elif runtime == "llamacpp": - open_url = f"http://127.0.0.1:{cfg.llamacpp_local_port}" else: open_url = f"http://127.0.0.1:{cfg.swarmui_local_port}" diff --git a/tests/test_access_card.py b/tests/test_access_card.py index e078d81..b25e8b7 100644 --- a/tests/test_access_card.py +++ b/tests/test_access_card.py @@ -5,7 +5,6 @@ class _Cfg: swarmui_local_port = 17801 llm_runtime = "ollama" ollama_local_port = 17811 - llamacpp_local_port = 17812 enable_swarmui = True diff --git a/tests/test_llamacpp_models.py b/tests/test_llamacpp_models.py deleted file mode 100644 index bd9f640..0000000 --- a/tests/test_llamacpp_models.py +++ /dev/null @@ -1,31 +0,0 @@ -from pathlib import Path - -from gpu_rent.llm_runtime import ( - gguf_filename_from_url, - parse_llamacpp_models, - write_llamacpp_models_preset, -) - - -def test_gguf_filename_from_url(): - url = ( - "https://huggingface.co/org/repo/resolve/main/" - "Qwen2.5-3B-Instruct-Q4_K_M.gguf" - ) - assert gguf_filename_from_url(url) == "Qwen2.5-3B-Instruct-Q4_K_M.gguf" - - -def test_write_and_parse_llamacpp_preset(tmp_path: Path): - path = tmp_path / "llamacpp-models.yaml" - write_llamacpp_models_preset(path, "light") - entries = parse_llamacpp_models(path) - assert len(entries) == 1 - assert entries[0].default is True - assert "Qwen2.5-3B" in entries[0].url - assert entries[0].url.startswith("https://") - - -def test_parse_llamacpp_empty(tmp_path: Path): - path = tmp_path / "llamacpp-models.yaml" - write_llamacpp_models_preset(path, "empty") - assert parse_llamacpp_models(path) == [] diff --git a/tests/test_llm_only.py b/tests/test_llm_only.py index ac8ad95..b3c5e8e 100644 --- a/tests/test_llm_only.py +++ b/tests/test_llm_only.py @@ -15,13 +15,12 @@ def test_parse_enable_swarmui_workload(monkeypatch): def test_tunnel_forwards_llm_only(monkeypatch): class Cfg: swarmui_local_port = 17801 - llm_runtime = "llamacpp" + llm_runtime = "ollama" enable_swarmui = False ollama_local_port = 17811 - llamacpp_local_port = 17812 monkeypatch.setattr("gpu_rent.tunnel.load_state", lambda: type("S", (), {"notes": {}})()) - assert tunnel_forwards(Cfg()) == [(17812, 8080)] + assert tunnel_forwards(Cfg()) == [(17811, 11434)] def test_access_links_llm_only(monkeypatch): @@ -29,10 +28,9 @@ def test_access_links_llm_only(monkeypatch): class Cfg: swarmui_local_port = 17801 - llm_runtime = "llamacpp" + llm_runtime = "ollama" enable_swarmui = False ollama_local_port = 17811 - llamacpp_local_port = 17812 monkeypatch.setattr( "gpu_rent.access_card.load_state", @@ -40,5 +38,5 @@ def test_access_links_llm_only(monkeypatch): ) labels = [x.label for x in collect_access_links(Cfg(), tunneled=True)] assert "SwarmUI UI" not in labels - assert "llama.cpp" in labels + assert "Ollama API" in labels assert mcp_snippet_lines(Cfg())[0].startswith("#") diff --git a/tests/test_llm_runtime.py b/tests/test_llm_runtime.py index 3dde913..8e27efd 100644 --- a/tests/test_llm_runtime.py +++ b/tests/test_llm_runtime.py @@ -5,10 +5,7 @@ import pytest from gpu_rent.llm_runtime import ( decide_runtime, normalize_runtime, - parse_llamacpp_models, parse_ollama_models, - pick_llamacpp_linux_asset_url, - remap_llamacpp_url, write_ollama_models_preset, ) @@ -16,22 +13,20 @@ from gpu_rent.llm_runtime import ( def test_normalize_runtime(): assert normalize_runtime(None) == "none" assert normalize_runtime("OLLAMA") == "ollama" - assert normalize_runtime("llama-cpp") == "llamacpp" + with pytest.raises(ValueError): + normalize_runtime("llamacpp") with pytest.raises(ValueError): normalize_runtime("foo") def test_decide_runtime_flags_win(): assert ( - decide_runtime(flag=None, ollama_flag=True, llamacpp_flag=False, from_config="none") - == "ollama" + decide_runtime(flag=None, ollama_flag=True, from_config="none") == "ollama" ) assert ( - decide_runtime(flag="llamacpp", ollama_flag=False, llamacpp_flag=False, from_config="ollama") - == "llamacpp" + decide_runtime(flag="ollama", ollama_flag=False, from_config="none") == "ollama" ) - with pytest.raises(ValueError): - decide_runtime(flag=None, ollama_flag=True, llamacpp_flag=True, from_config="none") + assert decide_runtime(flag=None, ollama_flag=False, from_config="ollama") == "ollama" def test_parse_ollama_models(tmp_path: Path): @@ -60,66 +55,3 @@ def test_write_preset(tmp_path: Path): write_ollama_models_preset(path, "recommended") entries = parse_ollama_models(path) assert entries[0].name == "huihui_ai/qwen2.5-vl-abliterated:7b" - - -def test_write_llamacpp_preset_includes_mmproj(tmp_path: Path): - from gpu_rent.llm_runtime import write_llamacpp_models_preset - - path = tmp_path / "lc.yaml" - write_llamacpp_models_preset(path, "recommended") - entries = parse_llamacpp_models(path) - assert len(entries) == 1 - assert "VL" in entries[0].url or "vl" in entries[0].url.lower() - assert entries[0].mmproj_url - assert "mmproj" in entries[0].mmproj_url - - -def test_remap_dead_bartowski_abliterate_url(tmp_path: Path): - dead = ( - "https://huggingface.co/bartowski/huihui-ai_Qwen2.5-7B-Instruct-abliterated-GGUF/" - "resolve/main/huihui-ai_Qwen2.5-7B-Instruct-abliterated-Q4_K_M.gguf" - ) - fixed = remap_llamacpp_url(dead) - assert "RichardErkhov" in fixed - assert "Q4_K_M.gguf" in fixed - path = tmp_path / "lc.yaml" - path.write_text(f"models:\n - url: {dead}\n default: true\n", encoding="utf-8") - entries = parse_llamacpp_models(path) - assert len(entries) == 1 - assert entries[0].url == fixed - - -def test_pick_llamacpp_linux_asset_skips_windows_cuda(): - assets = [ - { - "name": "cudart-llama-bin-win-cuda-12.4-x64.zip", - "browser_download_url": "https://example/cudart-win.zip", - }, - { - "name": "llama-b10545-bin-win-cuda-12.4-x64.zip", - "browser_download_url": "https://example/win-cuda.zip", - }, - { - "name": "llama-b10545-bin-ubuntu-x64.tar.gz", - "browser_download_url": "https://example/ubuntu-cpu.tar.gz", - }, - { - "name": "llama-b10545-bin-ubuntu-vulkan-x64.tar.gz", - "browser_download_url": "https://example/ubuntu-vulkan.tar.gz", - }, - ] - assert pick_llamacpp_linux_asset_url(assets) == "https://example/ubuntu-vulkan.tar.gz" - - -def test_pick_llamacpp_linux_asset_prefers_ubuntu_cuda(): - assets = [ - { - "name": "llama-b1-bin-ubuntu-vulkan-x64.tar.gz", - "browser_download_url": "https://example/vulkan.tar.gz", - }, - { - "name": "llama-b1-bin-ubuntu-cuda-12.4-x64.tar.gz", - "browser_download_url": "https://example/cuda.tar.gz", - }, - ] - assert pick_llamacpp_linux_asset_url(assets) == "https://example/cuda.tar.gz" diff --git a/tests/test_manifests.py b/tests/test_manifests.py index 6f6821e..e2994c5 100644 --- a/tests/test_manifests.py +++ b/tests/test_manifests.py @@ -53,7 +53,6 @@ def test_extensions_requires_ollama(tmp_path: Path): assert repos[1].requires == "none" assert repo_matches_runtime(repos[0], "ollama") assert not repo_matches_runtime(repos[0], "none") - assert not repo_matches_runtime(repos[0], "llamacpp") assert repo_matches_runtime(repos[1], "none") assert repo_matches_runtime(repos[1], "ollama") @@ -67,7 +66,6 @@ def test_extensions_requires_any_llm(tmp_path: Path): repo = parse_extensions(path)[0] assert repo.requires == "any-llm" assert repo_matches_runtime(repo, "ollama") - assert repo_matches_runtime(repo, "llamacpp") assert not repo_matches_runtime(repo, "none") diff --git a/tests/test_ready_snapshot.py b/tests/test_ready_snapshot.py index 369033d..5bc5f0b 100644 --- a/tests/test_ready_snapshot.py +++ b/tests/test_ready_snapshot.py @@ -8,7 +8,6 @@ class _Cfg: notify_ready = True llm_runtime = "none" ollama_local_port = 17811 - llamacpp_local_port = 17812 def test_ensure_boot_snapshot_skips_existing(): diff --git a/tests/test_remote_llm_env.py b/tests/test_remote_llm_env.py index 54974e4..74e0a78 100644 --- a/tests/test_remote_llm_env.py +++ b/tests/test_remote_llm_env.py @@ -1,27 +1,21 @@ from types import SimpleNamespace -from gpu_rent.provision import _LLAMACPP_INSTALL_ENV, _remote_llm_env +from gpu_rent.provision import _OLLAMA_INSTALL_ENV, _remote_llm_env -def test_remote_llm_env_forwards_llamacpp_vars(monkeypatch): - monkeypatch.setenv("LLAMACPP_TAG", "b10545") - monkeypatch.setenv("LLAMACPP_BUILD_CUDA", "1") - monkeypatch.setenv("LLAMACPP_NGL", "40") - monkeypatch.setenv("LLAMACPP_ASSET_URL", "https://example/a.tar.gz") - monkeypatch.delenv("LLAMACPP_SHA256", raising=False) +def test_remote_llm_env_forwards_ollama_vars(monkeypatch): + monkeypatch.setenv("OLLAMA_VERSION", "0.6.5") + monkeypatch.setenv("OLLAMA_SHA256", "abc123") cfg = SimpleNamespace(ssh_user="ubuntu") - env = _remote_llm_env(cfg, *_LLAMACPP_INSTALL_ENV) + env = _remote_llm_env(cfg, *_OLLAMA_INSTALL_ENV) assert env["SWARM_USER"] == "ubuntu" - assert env["LLAMACPP_TAG"] == "b10545" - assert env["LLAMACPP_BUILD_CUDA"] == "1" - assert env["LLAMACPP_NGL"] == "40" - assert env["LLAMACPP_ASSET_URL"] == "https://example/a.tar.gz" - assert "LLAMACPP_SHA256" not in env + assert env["OLLAMA_VERSION"] == "0.6.5" + assert env["OLLAMA_SHA256"] == "abc123" def test_remote_llm_env_skips_empty(monkeypatch): - for key in _LLAMACPP_INSTALL_ENV: + for key in _OLLAMA_INSTALL_ENV: monkeypatch.delenv(key, raising=False) cfg = SimpleNamespace(ssh_user="ubuntu") - env = _remote_llm_env(cfg, *_LLAMACPP_INSTALL_ENV) + env = _remote_llm_env(cfg, *_OLLAMA_INSTALL_ENV) assert env == {"SWARM_USER": "ubuntu"} diff --git a/tests/test_server_plan.py b/tests/test_server_plan.py index c931a60..9c3bb4b 100644 --- a/tests/test_server_plan.py +++ b/tests/test_server_plan.py @@ -55,7 +55,7 @@ def test_prompt_server_plan_with_ranked(monkeypatch, tmp_path): def test_upsert_vars(tmp_path): path = tmp_path / "gpu-rent.vars" path.write_text("# c\nLLM_RUNTIME=none\n", encoding="utf-8") - upsert_vars(path, {"LLM_RUNTIME": "llamacpp", "DATA_VOLUME_SIZE_GB": "200"}) + upsert_vars(path, {"LLM_RUNTIME": "ollama", "DATA_VOLUME_SIZE_GB": "200"}) data = parse_vars_file(path) - assert data["LLM_RUNTIME"] == "llamacpp" + assert data["LLM_RUNTIME"] == "ollama" assert data["DATA_VOLUME_SIZE_GB"] == "200" diff --git a/tests/test_tunnel_watch.py b/tests/test_tunnel_watch.py index 93fef45..f4b56a4 100644 --- a/tests/test_tunnel_watch.py +++ b/tests/test_tunnel_watch.py @@ -32,7 +32,6 @@ def test_tunnel_forwards_swarm_only(monkeypatch): swarmui_local_port = 17801 llm_runtime = "none" ollama_local_port = 17811 - llamacpp_local_port = 17812 monkeypatch.setattr("gpu_rent.tunnel.load_state", lambda: type("S", (), {"notes": {}})()) assert tunnel_forwards(Cfg()) == [(17801, 7801)] @@ -43,7 +42,6 @@ def test_tunnel_forwards_prefers_cfg_over_stale_notes(monkeypatch): swarmui_local_port = 17801 llm_runtime = "none" ollama_local_port = 17811 - llamacpp_local_port = 17812 monkeypatch.setattr( "gpu_rent.tunnel.load_state", @@ -65,10 +63,10 @@ def test_resolve_llm_notes_only_when_cfg_none(monkeypatch): assert resolve_llm_runtime(Cfg()) == "ollama" class Cfg2: - llm_runtime = "llamacpp" + llm_runtime = "ollama" monkeypatch.setattr( "gpu_rent.access_card.load_state", - lambda: type("S", (), {"notes": {"llm_runtime": "ollama"}})(), + lambda: type("S", (), {"notes": {"llm_runtime": "none"}})(), ) - assert resolve_llm_runtime(Cfg2()) == "llamacpp" + assert resolve_llm_runtime(Cfg2()) == "ollama" diff --git a/tests/test_verify_stack.py b/tests/test_verify_stack.py index 129e764..bd29c57 100644 --- a/tests/test_verify_stack.py +++ b/tests/test_verify_stack.py @@ -7,19 +7,18 @@ class _Cfg: llm_runtime = "none" swarmui_local_port = 17801 ollama_local_port = 17811 - llamacpp_local_port = 17812 def test_expected_services_swarm_only(): - assert _expected_services(_Cfg()) == (True, False, False) + assert _expected_services(_Cfg()) == (True, False) def test_expected_services_llm_only(): class C: enable_swarmui = False - llm_runtime = "llamacpp" + llm_runtime = "ollama" - assert _expected_services(C()) == (False, False, True) + assert _expected_services(C()) == (False, True) def test_verify_stack_local_empty_when_nothing(): @@ -28,7 +27,6 @@ def test_verify_stack_local_empty_when_nothing(): llm_runtime = "none" swarmui_local_port = 17801 ollama_local_port = 17811 - llamacpp_local_port = 17812 logs: list[str] = [] assert verify_stack_local(C(), logs.append, timeout=0.1) == [] @@ -40,7 +38,6 @@ def test_verify_stack_local_fails_closed_port(monkeypatch): llm_runtime = "ollama" swarmui_local_port = 17801 ollama_local_port = 17999 - llamacpp_local_port = 17812 monkeypatch.setattr( "gpu_rent.ready._tcp_ok", lambda port, host="127.0.0.1", timeout=0.8: False @@ -157,7 +154,7 @@ def test_verify_gpu_env_llm_only_skips_torch_requirement(monkeypatch): class C: enable_swarmui = False - llm_runtime = "llamacpp" + llm_runtime = "ollama" payload = { "ok": True,