Ship Assistent 0.8.1: split modules and shared+personal vector memory.

Personal RAG never leaks into the shared store; retrieve merges shared plus the persona chain, with personal overwrite on kind+key.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Leonid Pershin
2026-08-22 00:38:54 +03:00
co-authored by Cursor
parent 6d8aaefc38
commit 880e2dbea2
21 changed files with 3946 additions and 2040 deletions
+118
View File
@@ -0,0 +1,118 @@
using System;
using System.Net.Http;
using System.Text;
using System.Threading.Tasks;
using Newtonsoft.Json.Linq;
using SwarmUI.Accounts;
using SwarmUI.Utils;
namespace Mrleo1nid.SwarmAssistent;
/// <summary>VRAM handover between Ollama and the image backend: park the chat model before
/// Generate, warm it again once the user is back in the chat. Embed / memory models are never
/// parked — they are tiny and reloading them stalls every retrieve.</summary>
public partial class SwarmAssistentExtension
{
const string WarmKeepAlive = "15m";
/// <summary>Unloads the chat model from VRAM (<c>keep_alive: 0</c>) so Krea 2 gets the whole GPU.</summary>
public async Task<JObject> AssistentParkLlm(Session session, string baseUrl, string model)
{
string root = NormalizeBaseUrl(baseUrl);
string name = (model ?? "").Trim();
if (string.IsNullOrWhiteSpace(name))
{
return new JObject { ["error"] = "model is required" };
}
if (LooksLikeEmbedModel(name))
{
return new JObject { ["success"] = true, ["parked"] = false, ["skipped"] = "memory model — never parked" };
}
JObject generate = new()
{
["model"] = name,
["prompt"] = "",
["stream"] = false,
["keep_alive"] = 0,
};
(bool ok, string body) = await PostOllamaJson(root, "/api/generate", generate);
if (!ok)
{
// Older Ollama builds only unload through /api/chat.
JObject chat = new()
{
["model"] = name,
["messages"] = new JArray(),
["stream"] = false,
["keep_alive"] = 0,
};
(ok, body) = await PostOllamaJson(root, "/api/chat", chat);
}
if (!ok)
{
Logs.Debug($"AssistentParkLlm {name}: {Clip(body, 200)}");
return new JObject { ["success"] = true, ["parked"] = false, ["note"] = Clip(body, 200) };
}
return new JObject { ["success"] = true, ["parked"] = true, ["model"] = name, ["base_url"] = root };
}
/// <summary>Single-token chat so the model is resident again by the time the user types.</summary>
public async Task<JObject> AssistentWarmLlm(Session session, string baseUrl, string model, string persona = null)
{
string root = NormalizeBaseUrl(baseUrl);
string name = (model ?? "").Trim();
if (string.IsNullOrWhiteSpace(name))
{
return new JObject { ["error"] = "model is required" };
}
if (LooksLikeEmbedModel(name))
{
return new JObject { ["success"] = true, ["warmed"] = false, ["skipped"] = "memory model" };
}
int numCtx = CfgInt("num_ctx", DefaultNumCtxFallback);
JObject payload = new()
{
["model"] = name,
["stream"] = false,
["messages"] = new JArray
{
new JObject { ["role"] = "user", ["content"] = "ok" },
},
["options"] = new JObject
{
["num_ctx"] = numCtx,
["num_predict"] = 1,
},
["keep_alive"] = WarmKeepAlive,
};
(bool ok, string body) = await PostOllamaJson(root, "/api/chat", payload);
if (!ok)
{
Logs.Debug($"AssistentWarmLlm {name}: {Clip(body, 200)}");
return new JObject { ["success"] = true, ["warmed"] = false, ["note"] = Clip(body, 200) };
}
return new JObject
{
["success"] = true,
["warmed"] = true,
["model"] = name,
["num_ctx"] = numCtx,
["keep_alive"] = WarmKeepAlive,
};
}
static async Task<(bool ok, string body)> PostOllamaJson(string root, string route, JObject payload)
{
try
{
using StringContent content = new(payload.ToString(Newtonsoft.Json.Formatting.None), Encoding.UTF8, "application/json");
using HttpResponseMessage resp = await HttpClient.PostAsync($"{root}{route}", content);
string body = await resp.Content.ReadAsStringAsync();
return (resp.IsSuccessStatusCode, resp.IsSuccessStatusCode ? body : $"HTTP {(int)resp.StatusCode}: {body}");
}
catch (Exception ex)
{
return (false, ex.Message);
}
}
}