From 6e3f94bc15b4b191ff62b81f601e7b44e6a00126 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Wed, 23 Sep 2026 22:33:15 -0700 Subject: [PATCH 01/70] feat(engine-manager): support llama.cpp SSE pulls Signed-off-by: Sherief Farouk --- services/nvpair-engine-manager/pull.go | 131 ++++++++++++++++- services/nvpair-engine-manager/pull_test.go | 132 ++++++++++++++++++ services/nvpair-engine-manager/registry.go | 13 ++ .../nvpair-engine-manager/registry_test.go | 23 +++ 4 files changed, 293 insertions(+), 6 deletions(-) diff --git a/services/nvpair-engine-manager/pull.go b/services/nvpair-engine-manager/pull.go index 6ffd8cca..ecc7270b 100644 --- a/services/nvpair-engine-manager/pull.go +++ b/services/nvpair-engine-manager/pull.go @@ -14,10 +14,10 @@ package main // Ollama's /api/pull streams newline-delimited JSON status objects // ({"status":...,"total":N,"completed":M}); each line maps to a progress event, // coalesced so only changes in stage/percent are emitted (a single layer streams -// many byte-progress lines at the same rendered percent). CLI-driven pulls (LM -// Studio's `lms get`) don't expose structured line progress here, so they emit a -// single "pulling" marker and return the final result — the security/trust -// boundary and result contract are identical. +// many byte-progress lines at the same rendered percent). llama.cpp instead +// acknowledges POST /models immediately and reports completion on /models/sse; +// its manifest opts into that named adapter. CLI-driven pulls (LM Studio's +// `lms get`) emit one "pulling" marker and return the final result. import ( "bufio" @@ -31,8 +31,13 @@ import ( "strings" ) -// pullModelAction is the manifest action name every engine uses for model pulls. -const pullModelAction = "pull_model" +const ( + // pullModelAction is the manifest action name every engine uses for model pulls. + pullModelAction = "pull_model" + // pullProgressProtocolLlamaCPPModelsSSE names the pinned llama.cpp router + // protocol: subscribe first, POST /models, then await a matching terminal SSE. + pullProgressProtocolLlamaCPPModelsSSE = "llamacpp-models-sse" +) // modelFromParams extracts a human-readable model name from an engine:action // pull_model params object, preferring Ollama's "name" body key then the generic @@ -91,6 +96,9 @@ func (e *Executor) PullModelStream(ctx context.Context, engine, model string, pa if !running { return nil, fmt.Errorf("engine %q is not running", engine) } + if act.ProgressProtocol == pullProgressProtocolLlamaCPPModelsSSE { + return e.pullModelLlamaCPPSSE(ctx, engine, model, act, port, params) + } path, err := resolvePlaceholders(act.HTTP.Path, map[string]string{"port": strconv.Itoa(port)}) if err != nil { return nil, err @@ -140,6 +148,117 @@ func (e *Executor) PullModelStream(ctx context.Context, engine, model string, pa return last, nil } +// pullModelLlamaCPPSSE runs llama.cpp's asynchronous router download protocol. +// The SSE response must be open before POST /models because terminal events are +// one-shot broadcasts; subscribing afterward can miss a fast completion. +func (e *Executor) pullModelLlamaCPPSSE(ctx context.Context, engine, model string, act Action, port int, params json.RawMessage) (json.RawMessage, error) { + if strings.TrimSpace(model) == "" { + return nil, fmt.Errorf("pull model is required for %s", pullProgressProtocolLlamaCPPModelsSSE) + } + baseURL := fmt.Sprintf("http://127.0.0.1:%d", port) + sseReq, err := http.NewRequestWithContext(ctx, http.MethodGet, baseURL+"/models/sse", nil) + if err != nil { + return nil, err + } + sseReq.Header.Set("Accept", "text/event-stream") + sseResp, err := e.client.Do(sseReq) + if err != nil { + return nil, fmt.Errorf("pull %q: subscribe to model progress: %w", model, err) + } + defer sseResp.Body.Close() + if sseResp.StatusCode < 200 || sseResp.StatusCode >= 300 { + data, _ := io.ReadAll(io.LimitReader(sseResp.Body, 64*1024)) + return nil, fmt.Errorf("pull %q: progress stream returned HTTP %d: %s", model, sseResp.StatusCode, strings.TrimSpace(string(data))) + } + + path, err := resolvePlaceholders(act.HTTP.Path, map[string]string{"port": strconv.Itoa(port)}) + if err != nil { + return nil, err + } + startReq, err := http.NewRequestWithContext(ctx, strings.ToUpper(act.HTTP.Method), baseURL+path, bytes.NewReader(params)) + if err != nil { + return nil, err + } + startReq.Header.Set("Content-Type", "application/json") + startResp, err := e.client.Do(startReq) + if err != nil { + return nil, fmt.Errorf("pull %q: start download: %w", model, err) + } + startData, readErr := io.ReadAll(io.LimitReader(startResp.Body, 64*1024)) + startResp.Body.Close() + if readErr != nil { + return nil, fmt.Errorf("pull %q: read start response: %w", model, readErr) + } + if startResp.StatusCode < 200 || startResp.StatusCode >= 300 { + return nil, fmt.Errorf("pull %q: engine returned HTTP %d: %s", model, startResp.StatusCode, strings.TrimSpace(string(startData))) + } + var started struct { + Success bool `json:"success"` + } + if err := json.Unmarshal(startData, &started); err != nil || !started.Success { + return nil, fmt.Errorf("pull %q: engine did not accept the download", model) + } + + lastPct := -1 + sc := bufio.NewScanner(sseResp.Body) + sc.Buffer(make([]byte, 0, 64*1024), 1<<20) + for sc.Scan() { + line := bytes.TrimSpace(sc.Bytes()) + if !bytes.HasPrefix(line, []byte("data:")) { + continue + } + var event llamaCPPModelsEvent + if err := json.Unmarshal(bytes.TrimSpace(bytes.TrimPrefix(line, []byte("data:"))), &event); err != nil || event.Model != model { + continue + } + switch event.Event { + case "download_progress": + pct := event.percent() + if pct != lastPct { + lastPct = pct + e.emitPullProgress(ProgressEvent{Engine: engine, Op: "pull", Stage: "downloading", Percent: pct, Message: model}) + } + case "download_finished": + e.emitPullProgress(ProgressEvent{Engine: engine, Op: "pull", Stage: "success", Percent: 100, Message: model}) + return json.RawMessage(startData), nil + case "download_failed": + return nil, fmt.Errorf("pull %q: llama.cpp reported download failure", model) + } + } + if err := sc.Err(); err != nil { + return nil, fmt.Errorf("pull %q: progress stream: %w", model, err) + } + return nil, fmt.Errorf("pull %q: progress stream ended before completion", model) +} + +type llamaCPPModelsEvent struct { + Model string `json:"model"` + Event string `json:"event"` + Data struct { + Progress map[string]struct { + Done int64 `json:"done"` + Total int64 `json:"total"` + } `json:"progress"` + } `json:"data"` +} + +func (e llamaCPPModelsEvent) percent() int { + var done, total int64 + for _, file := range e.Data.Progress { + if file.Total <= 0 { + continue + } + total += file.Total + if file.Done > 0 { + done += min(file.Done, file.Total) + } + } + if total == 0 { + return 0 + } + return int(done * 100 / total) +} + // engineDisplayName returns the manifest display name for user-facing copy. func (e *Executor) engineDisplayName(engine string) string { st, err := e.state(engine) diff --git a/services/nvpair-engine-manager/pull_test.go b/services/nvpair-engine-manager/pull_test.go index 76f48197..ed94b6f8 100644 --- a/services/nvpair-engine-manager/pull_test.go +++ b/services/nvpair-engine-manager/pull_test.go @@ -7,9 +7,14 @@ import ( "bytes" "context" "encoding/json" + "fmt" + "net" + "net/http" "net/http/httptest" + "strconv" "strings" "sync" + "sync/atomic" "testing" ) @@ -59,6 +64,133 @@ func TestModelFromParams(t *testing.T) { } } +func TestLlamaCPPModelsEventPercentAggregatesFiles(t *testing.T) { + var event llamaCPPModelsEvent + err := json.Unmarshal([]byte(`{ + "model":"owner/repo:Q4_K_M", + "event":"download_progress", + "data":{"progress":{ + "model.gguf":{"done":75,"total":100}, + "mmproj.gguf":{"done":25,"total":100} + }} + }`), &event) + if err != nil { + t.Fatalf("decode event: %v", err) + } + if got := event.percent(); got != 50 { + t.Fatalf("aggregate percent = %d, want 50", got) + } +} + +func TestPullModelLlamaCPPSSESubscribesBeforeStarting(t *testing.T) { + const model = "owner/repo:Q4_K_M" + subscribed := make(chan struct{}) + started := make(chan struct{}) + var postBeforeSubscribe atomic.Bool + mux := http.NewServeMux() + mux.HandleFunc("/models/sse", func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "text/event-stream") + close(subscribed) + _, _ = fmt.Fprint(w, ": ready\n\n") + w.(http.Flusher).Flush() + select { + case <-started: + case <-r.Context().Done(): + return + } + write := func(event map[string]any) { + data, err := json.Marshal(event) + if err != nil { + t.Errorf("encode SSE event: %v", err) + return + } + _, _ = fmt.Fprintf(w, "data: %s\n\n", data) + w.(http.Flusher).Flush() + } + write(map[string]any{"model": "other/model", "event": "download_finished", "data": map[string]any{}}) + write(map[string]any{ + "model": model, + "event": "download_progress", + "data": map[string]any{"progress": map[string]any{ + "model.gguf": map[string]int64{"done": 75, "total": 100}, + "mmproj.gguf": map[string]int64{"done": 25, "total": 100}, + }}, + }) + write(map[string]any{"model": model, "event": "download_finished", "data": map[string]any{}}) + }) + mux.HandleFunc("/models", func(w http.ResponseWriter, r *http.Request) { + select { + case <-subscribed: + default: + postBeforeSubscribe.Store(true) + } + var body struct { + Model string `json:"model"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil || body.Model != model { + http.Error(w, "invalid model", http.StatusBadRequest) + return + } + close(started) + w.Header().Set("Content-Type", "application/json") + _, _ = fmt.Fprint(w, `{"success":true}`) + }) + server := httptest.NewServer(mux) + defer server.Close() + + _, portText, err := net.SplitHostPort(server.Listener.Addr().String()) + if err != nil { + t.Fatalf("split test server address: %v", err) + } + port, err := strconv.Atoi(portText) + if err != nil { + t.Fatalf("parse test server port: %v", err) + } + m := testEngineManifest(fakeEngineBin) + m.Actions[pullModelAction] = Action{ + HTTP: &ActionHTTP{Method: http.MethodPost, Path: "/models"}, + ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE, + } + reg := NewRegistry() + reg.engines[m.Engine] = m + ex := NewExecutor(reg, NewReporter(nil), nil, t.TempDir()) + st, err := ex.state(m.Engine) + if err != nil { + t.Fatalf("resolve engine state: %v", err) + } + st.running = true + st.port = port + progress, cancel := ex.progress.subscribe(m.Engine) + defer cancel() + + result, err := ex.PullModelStream(context.Background(), m.Engine, model, json.RawMessage(`{"model":"`+model+`"}`)) + if err != nil { + t.Fatalf("pull model: %v", err) + } + var response struct { + Success bool `json:"success"` + } + if err := json.Unmarshal(result, &response); err != nil || !response.Success { + t.Fatalf("pull result = %s, error %v", result, err) + } + if postBeforeSubscribe.Load() { + t.Fatal("download POST arrived before the SSE subscription was open") + } + var events []ProgressEvent + for len(progress) > 0 { + events = append(events, <-progress) + } + if len(events) != 2 { + t.Fatalf("progress events = %+v, want downloading and success", events) + } + if events[0].Stage != "downloading" || events[0].Percent != 50 { + t.Fatalf("download event = %+v, want 50%%", events[0]) + } + if events[1].Stage != "success" || events[1].Percent != 100 { + t.Fatalf("terminal event = %+v, want success at 100%%", events[1]) + } +} + // TestActionPullModelStreamsProgress verifies that engine:action with action // "pull_model" is routed through the streaming pull path, so a local pull // emits live engine:pull-progress notifications (with computed percentages) and diff --git a/services/nvpair-engine-manager/registry.go b/services/nvpair-engine-manager/registry.go index b6a79ad6..1ecd66bf 100644 --- a/services/nvpair-engine-manager/registry.go +++ b/services/nvpair-engine-manager/registry.go @@ -193,6 +193,11 @@ type Action struct { HTTP *ActionHTTP `json:"http,omitempty"` Cmd []string `json:"cmd,omitempty"` RemovePath *ActionRemovePath `json:"remove_path,omitempty"` + // ProgressProtocol selects a narrowly defined streaming adapter for an + // asynchronous pull HTTP action. Empty keeps the ordinary response-stream + // behavior; named protocols are validated so a typo cannot silently fall + // back to the wrong completion semantics. + ProgressProtocol string `json:"progress_protocol,omitempty"` // ModelResolution, when set, expands or resolves the model param: // - "lms-get" on Cmd actions: try as-given → Hub id → Hugging Face URL. // - "lms-disk-path" on RemovePath actions: map logical ids to on-disk @@ -702,6 +707,14 @@ func (a *Action) validate(name string) error { if hasHTTP && (strings.TrimSpace(a.HTTP.Method) == "" || strings.TrimSpace(a.HTTP.Path) == "") { return fmt.Errorf("action %q: http.method and http.path are required", name) } + if a.ProgressProtocol != "" { + if name != pullModelAction || !hasHTTP { + return fmt.Errorf("action %q: progress_protocol requires the HTTP pull_model action", name) + } + if a.ProgressProtocol != pullProgressProtocolLlamaCPPModelsSSE { + return fmt.Errorf("action %q: unsupported progress_protocol %q", name, a.ProgressProtocol) + } + } if a.Result != nil && (strings.TrimSpace(a.Result.Array) == "" || strings.TrimSpace(a.Result.Field) == "") { return fmt.Errorf("action %q: result.array and result.field are required when result is set", name) } diff --git a/services/nvpair-engine-manager/registry_test.go b/services/nvpair-engine-manager/registry_test.go index 2ae2c5c6..90c650a3 100644 --- a/services/nvpair-engine-manager/registry_test.go +++ b/services/nvpair-engine-manager/registry_test.go @@ -69,6 +69,17 @@ func TestValidateAcceptsCommandModeAndCmdAction(t *testing.T) { } } +func TestValidateAcceptsLlamaCPPModelPullProtocol(t *testing.T) { + m := validManifest() + m.Actions[pullModelAction] = Action{ + HTTP: &ActionHTTP{Method: "POST", Path: "/models"}, + ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE, + } + if err := m.Validate(); err != nil { + t.Fatalf("llama.cpp model pull protocol rejected: %v", err) + } +} + func TestValidateAcceptsUnpinnedFetch(t *testing.T) { m := validManifest() p := m.Platforms["linux/amd64"] @@ -139,6 +150,18 @@ func TestValidateRejects(t *testing.T) { {"action missing method", func(m *Manifest) { m.Actions = map[string]Action{"x": {HTTP: &ActionHTTP{Path: "/p"}}} }, "http.method and http.path"}, + {"unknown progress protocol", func(m *Manifest) { + m.Actions[pullModelAction] = Action{ + HTTP: &ActionHTTP{Method: "POST", Path: "/models"}, + ProgressProtocol: "unknown", + } + }, "unsupported progress_protocol"}, + {"progress protocol on other action", func(m *Manifest) { + m.Actions["list_models"] = Action{ + HTTP: &ActionHTTP{Method: "GET", Path: "/models"}, + ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE, + } + }, "requires the HTTP pull_model action"}, } for _, tc := range cases { t.Run(tc.name, func(t *testing.T) { From 830d755c6d10caba7ebfce54cf9a32a1d5f4f073 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Wed, 23 Sep 2026 23:04:16 -0700 Subject: [PATCH 02/70] feat(engine-manager): match nested model status Signed-off-by: Sherief Farouk --- services/nvpair-engine-manager/models.go | 29 +++++++++++++++---- services/nvpair-engine-manager/models_test.go | 6 ++++ services/nvpair-engine-manager/registry.go | 11 +++++-- .../nvpair-engine-manager/registry_test.go | 25 ++++++++++++++++ 4 files changed, 62 insertions(+), 9 deletions(-) diff --git a/services/nvpair-engine-manager/models.go b/services/nvpair-engine-manager/models.go index de89064e..31db17aa 100644 --- a/services/nvpair-engine-manager/models.go +++ b/services/nvpair-engine-manager/models.go @@ -8,6 +8,7 @@ import ( "context" "encoding/json" "log/slog" + "strings" "sync" "time" ) @@ -217,13 +218,13 @@ func extractStringsResult(raw json.RawMessage, spec *ActionResult) ([]string, bo } // matchRow reports whether an element passes an ActionResult row filter. -// With Match.In set, Match.Field must decode as a JSON string equal to one of -// In. With Match.Nonempty set, Match.Field must decode as a JSON array with -// length > 0 (LM Studio /api/v1/models loaded_instances). A missing or -// wrong-typed field fails the match, so a row we cannot classify is excluded -// rather than counted as loaded. +// Match.Field is a validated dot-separated object path. With Match.In set, the +// resolved value must decode as a JSON string equal to one of In. With +// Match.Nonempty set, it must decode as a JSON array with length > 0 (LM Studio +// /api/v1/models loaded_instances). A missing or wrong-typed path fails the +// match, so a row we cannot classify is excluded rather than counted as loaded. func matchRow(el map[string]json.RawMessage, m *ResultMatch) bool { - fv, ok := el[m.Field] + fv, ok := resolveObjectPath(el, m.Field) if !ok { return false } @@ -245,3 +246,19 @@ func matchRow(el map[string]json.RawMessage, m *ResultMatch) bool { } return false } + +func resolveObjectPath(obj map[string]json.RawMessage, path string) (json.RawMessage, bool) { + parts := strings.Split(path, ".") + value, ok := obj[parts[0]] + for _, part := range parts[1:] { + if !ok { + return nil, false + } + var nested map[string]json.RawMessage + if err := json.Unmarshal(value, &nested); err != nil { + return nil, false + } + value, ok = nested[part] + } + return value, ok +} diff --git a/services/nvpair-engine-manager/models_test.go b/services/nvpair-engine-manager/models_test.go index a9794f80..96936bee 100644 --- a/services/nvpair-engine-manager/models_test.go +++ b/services/nvpair-engine-manager/models_test.go @@ -68,6 +68,12 @@ func TestExtractStrings(t *testing.T) { spec: &ActionResult{Array: "data", Field: "id", Match: &ResultMatch{Field: "state", In: []string{"loaded"}}}, want: []string{"a"}, }, + { + name: "nested string-in match excludes missing and malformed paths", + raw: `{"data":[{"id":"a","status":{"value":"loaded"}},{"id":"b","status":{"value":"not-loaded"}},{"id":"c","status":{}},{"id":"d","status":"loaded"}]}`, + spec: &ActionResult{Array: "data", Field: "id", Match: &ResultMatch{Field: "status.value", In: []string{"loaded"}}}, + want: []string{"a"}, + }, { name: "match with no accepted values yields nothing", raw: `{"data":[{"id":"a","state":"loaded"}]}`, diff --git a/services/nvpair-engine-manager/registry.go b/services/nvpair-engine-manager/registry.go index 1ecd66bf..720e3dcb 100644 --- a/services/nvpair-engine-manager/registry.go +++ b/services/nvpair-engine-manager/registry.go @@ -40,6 +40,7 @@ var allowedPlaceholders = map[string]bool{ } var placeholderRe = regexp.MustCompile(`\{([a-zA-Z_][a-zA-Z0-9_]*)\}`) +var resultMatchFieldPathRe = regexp.MustCompile(`^[a-zA-Z_][a-zA-Z0-9_-]*(?:\.[a-zA-Z_][a-zA-Z0-9_-]*)*$`) // engineNameRe restricts engine names to a safe charset — the name is // used as a filesystem path component (the per-engine install dir), so @@ -233,9 +234,10 @@ type ActionResult struct { // ResultMatch is the optional row filter on an ActionResult. Exactly one of // In or Nonempty must be set: -// - In: keep the element when Field (decoded as a JSON string) equals one of In. -// - Nonempty: keep the element when Field is a JSON array with length > 0 -// (LM Studio's /api/v1/models models[].loaded_instances). +// - In: keep the element when the dot-separated object path Field (decoded as +// a JSON string) equals one of In. +// - Nonempty: keep the element when Field resolves to a JSON array with length +// > 0 (LM Studio's /api/v1/models models[].loaded_instances). type ResultMatch struct { Field string `json:"field"` // element field to test, e.g. "state" / "loaded_instances" In []string `json:"in,omitempty"` // accepted string values, e.g. ["loaded"] @@ -723,6 +725,9 @@ func (a *Action) validate(name string) error { if strings.TrimSpace(m.Field) == "" { return fmt.Errorf("action %q: result.match.field is required when result.match is set", name) } + if !resultMatchFieldPathRe.MatchString(m.Field) { + return fmt.Errorf("action %q: result.match.field %q is not a valid object path", name, m.Field) + } hasIn := len(m.In) > 0 if hasIn == m.Nonempty { return fmt.Errorf("action %q: result.match requires exactly one of a non-empty in or nonempty=true", name) diff --git a/services/nvpair-engine-manager/registry_test.go b/services/nvpair-engine-manager/registry_test.go index 90c650a3..29102c99 100644 --- a/services/nvpair-engine-manager/registry_test.go +++ b/services/nvpair-engine-manager/registry_test.go @@ -80,6 +80,21 @@ func TestValidateAcceptsLlamaCPPModelPullProtocol(t *testing.T) { } } +func TestValidateAcceptsNestedResultMatch(t *testing.T) { + m := validManifest() + m.Actions["list_models"] = Action{ + HTTP: &ActionHTTP{Method: "GET", Path: "/models"}, + Result: &ActionResult{ + Array: "data", + Field: "id", + Match: &ResultMatch{Field: "status.value", In: []string{"loaded"}}, + }, + } + if err := m.Validate(); err != nil { + t.Fatalf("nested result match rejected: %v", err) + } +} + func TestValidateAcceptsUnpinnedFetch(t *testing.T) { m := validManifest() p := m.Platforms["linux/amd64"] @@ -162,6 +177,16 @@ func TestValidateRejects(t *testing.T) { ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE, } }, "requires the HTTP pull_model action"}, + {"invalid match field path", func(m *Manifest) { + m.Actions["list_models"] = Action{ + HTTP: &ActionHTTP{Method: "GET", Path: "/models"}, + Result: &ActionResult{ + Array: "models", + Field: "id", + Match: &ResultMatch{Field: "status..value", In: []string{"loaded"}}, + }, + } + }, "is not a valid object path"}, } for _, tc := range cases { t.Run(tc.name, func(t *testing.T) { From c3884ecbd2ab5ff92133265d656be7d73a9e42a2 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Wed, 23 Sep 2026 23:18:13 -0700 Subject: [PATCH 03/70] feat(proxy): remap model-list upstream paths Signed-off-by: Sherief Farouk --- services/nvpair-proxy/engines.go | 25 +++++++++++++++++++------ services/nvpair-proxy/failover_test.go | 26 ++++++++++++++++++++++++++ services/nvpair-proxy/proxy.go | 20 ++++++++++++-------- 3 files changed, 57 insertions(+), 14 deletions(-) diff --git a/services/nvpair-proxy/engines.go b/services/nvpair-proxy/engines.go index 71201a30..2a5054b2 100644 --- a/services/nvpair-proxy/engines.go +++ b/services/nvpair-proxy/engines.go @@ -78,8 +78,9 @@ const ( // route is one classified request path. type route struct { - Path string - Role routeRole + Path string + UpstreamPath string + Role routeRole } // engineProfile is everything the proxy needs to front one engine. @@ -195,9 +196,9 @@ func engineNames() string { return strings.Join(names, ", ") } -// roleFor classifies a request. The bool reports whether the path is one this +// routeFor classifies a request. The bool reports whether the path is one this // engine handles specially; false means forward it verbatim. -func (p engineProfile) roleFor(method, path string) (routeRole, bool) { +func (p engineProfile) routeFor(method, path string) (route, bool) { for _, r := range p.Routes { // Keep scanning on a method mismatch rather than bailing: a path may // legitimately appear twice under different methods, and returning @@ -206,9 +207,21 @@ func (p engineProfile) roleFor(method, path string) (routeRole, bool) { if r.Path != path || r.Role.method() != method { continue } - return r.Role, true + return r, true + } + return route{}, false +} + +func (p engineProfile) roleFor(method, path string) (routeRole, bool) { + r, ok := p.routeFor(method, path) + return r.Role, ok +} + +func (r route) upstreamPath() string { + if r.UpstreamPath != "" { + return r.UpstreamPath } - return 0, false + return r.Path } // method is the HTTP method a role applies to. diff --git a/services/nvpair-proxy/failover_test.go b/services/nvpair-proxy/failover_test.go index 0d54b6fe..a1505226 100644 --- a/services/nvpair-proxy/failover_test.go +++ b/services/nvpair-proxy/failover_test.go @@ -536,6 +536,32 @@ func TestHandleHTTP_AggregatesOpenAIModelList(t *testing.T) { }) } +func TestHandleHTTP_ModelListRemapsUpstreamPath(t *testing.T) { + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodGet || r.URL.Path != "/models" || r.URL.RawQuery != "scope=all" { + t.Errorf("upstream request = %s %s?%s, want GET /models?scope=all", r.Method, r.URL.Path, r.URL.RawQuery) + } + _, _ = io.WriteString(w, `{"data":[{"id":"remapped"}]}`) + })) + defer upstream.Close() + + profile := lmstudioCase(t).profile + profile.Routes = []route{{ + Path: "/v1/models", + UpstreamPath: "/models", + Role: roleModelListOpenAIGET, + }} + disc := NewDiscovery() + disc.AddManual(nodeFor(t, "remapped", upstream.URL)) + rec := httptest.NewRecorder() + testProxy(profile, disc, profile.FacadePort).soleFacade(). + handleHTTP(rec, httptest.NewRequest(http.MethodGet, "/v1/models?scope=all", nil)) + + if rec.Code != http.StatusOK || !strings.Contains(rec.Body.String(), `"id":"remapped"`) { + t.Fatalf("response = %d %s, want remapped model list", rec.Code, rec.Body.String()) + } +} + func TestHandleHTTP_ModelListEmptyAndUnavailable(t *testing.T) { forEachEngine(t, func(t *testing.T, tc engineCase) { serveEmpty := func() *httptest.Server { diff --git a/services/nvpair-proxy/proxy.go b/services/nvpair-proxy/proxy.go index 929c3d67..52fb0f25 100644 --- a/services/nvpair-proxy/proxy.go +++ b/services/nvpair-proxy/proxy.go @@ -931,11 +931,11 @@ type modelListResult struct { // in candidate order, not completion order, so duplicate metadata is // deterministic while an unavailable peer cannot hide healthy inventories. // -// role selects the wire dialect: which array the upstream envelope carries, -// which field identifies a record, and how the federated response is shaped. -func (f *facade) serveModelList(w http.ResponseWriter, r *http.Request, role routeRole, candidates []candidate) (int, error) { +// matchedRoute selects the wire dialect and the optional upstream path. The +// client-facing path remains unchanged in telemetry and in the merged response. +func (f *facade) serveModelList(w http.ResponseWriter, r *http.Request, matchedRoute route, candidates []candidate) (int, error) { p := f.host - openAI := role == roleModelListOpenAIGET + openAI := matchedRoute.Role == roleModelListOpenAIGET writeJSON := func(status int, body []byte) { w.Header().Set("Content-Type", "application/json") w.Header().Set("X-Content-Type-Options", "nosniff") @@ -946,8 +946,12 @@ func (f *facade) serveModelList(w http.ResponseWriter, r *http.Request, role rou var wg sync.WaitGroup for i, cand := range candidates { target := *cand.url - target.Path = r.URL.Path - target.RawPath = r.URL.RawPath + target.Path = matchedRoute.upstreamPath() + if matchedRoute.UpstreamPath == "" { + target.RawPath = r.URL.RawPath + } else { + target.RawPath = "" + } target.RawQuery = r.URL.RawQuery upstream, err := http.NewRequestWithContext(r.Context(), http.MethodGet, target.String(), nil) if err != nil { @@ -1208,13 +1212,13 @@ func (f *facade) handleHTTP(w http.ResponseWriter, r *http.Request) { p.releaseReservation(held) held = reservation{} }() - if role, ok := f.profile.roleFor(r.Method, r.URL.Path); ok && role.isModelList() { + if matchedRoute, ok := f.profile.routeFor(r.Method, r.URL.Path); ok && matchedRoute.Role.isModelList() { if len(candidates) > 0 { _ = f.notify("proxy/request-started", RequestStartedEvent{ ID: reqID, Method: r.Method, Path: r.URL.Path, Target: "cluster", }) } - status, err := f.serveModelList(w, r, role, candidates) + status, err := f.serveModelList(w, r, matchedRoute, candidates) errText := "" if err != nil { errText = err.Error() From 5603daf8819ffc4bc596b0c051db9941dba7a666 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Wed, 23 Sep 2026 23:28:56 -0700 Subject: [PATCH 04/70] feat(engines): support opt-in proxy defaults Signed-off-by: Sherief Farouk --- services/nvpair-tui/ui/proxies.go | 12 +++--- services/nvpair-tui/ui/proxies_test.go | 24 +++++++++++ services/nvpair-ui-broker/engineproxy_test.go | 6 ++- services/nvpair-ui-broker/main.go | 11 ++++- services/shared/engines/engines.go | 42 +++++++++++++------ services/shared/engines/engines_test.go | 13 ++++++ 6 files changed, 88 insertions(+), 20 deletions(-) create mode 100644 services/nvpair-tui/ui/proxies_test.go diff --git a/services/nvpair-tui/ui/proxies.go b/services/nvpair-tui/ui/proxies.go index 42c36dca..49a3ca34 100644 --- a/services/nvpair-tui/ui/proxies.go +++ b/services/nvpair-tui/ui/proxies.go @@ -25,13 +25,13 @@ type proxyNode struct { Port int `json:"port"` } -// buildProxyEngines makes one tab per engine, in the shared table's order, so -// an engine added there appears here rather than being silently absent from -// this view. +// buildProxyEngines makes one tab per default-enabled facade, in the shared +// table's order. Opt-in engines appear only when the TUI passes an explicit +// selection to the broker and this view. func buildProxyEngines() []*proxyEngine { - all := engines.All() - out := make([]*proxyEngine, 0, len(all)) - for _, e := range all { + defaults := engines.ProxyDefaults() + out := make([]*proxyEngine, 0, len(defaults)) + for _, e := range defaults { out = append(out, &proxyEngine{label: e.DisplayName, prefix: e.ComponentName(), table: newTable(nil)}) } return out diff --git a/services/nvpair-tui/ui/proxies_test.go b/services/nvpair-tui/ui/proxies_test.go new file mode 100644 index 00000000..f0460fe9 --- /dev/null +++ b/services/nvpair-tui/ui/proxies_test.go @@ -0,0 +1,24 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package ui + +import ( + "testing" + + "nvpair-shared/engines" +) + +func TestBuildProxyEnginesUsesSharedDefaults(t *testing.T) { + got := buildProxyEngines() + want := engines.ProxyDefaults() + if len(got) != len(want) { + t.Fatalf("proxy tabs = %d, want %d", len(got), len(want)) + } + for i, engine := range want { + if got[i].label != engine.DisplayName || got[i].prefix != engine.ComponentName() { + t.Errorf("proxy tab %d = (%q, %q), want (%q, %q)", + i, got[i].label, got[i].prefix, engine.DisplayName, engine.ComponentName()) + } + } +} diff --git a/services/nvpair-ui-broker/engineproxy_test.go b/services/nvpair-ui-broker/engineproxy_test.go index 7319e55e..1c650063 100644 --- a/services/nvpair-ui-broker/engineproxy_test.go +++ b/services/nvpair-ui-broker/engineproxy_test.go @@ -161,13 +161,17 @@ func TestOwnershipDecidesTheOccupiedFacadeOutcome(t *testing.T) { // --proxy-engines selects which engines the one binary is started for. func TestParseProxyEngines(t *testing.T) { + defaults := defaultProxyEngineNames() + if len(defaults) != 2 || defaults[0] != "ollama" || defaults[1] != "lmstudio" { + t.Fatalf("defaultProxyEngineNames() = %v, want [ollama lmstudio]", defaults) + } for _, tc := range []struct { name string csv string want []string wantErr bool }{ - {name: "default is every engine", csv: "ollama,lmstudio", want: []string{"ollama", "lmstudio"}}, + {name: "default set", csv: "ollama,lmstudio", want: []string{"ollama", "lmstudio"}}, {name: "single engine", csv: "lmstudio", want: []string{"lmstudio"}}, {name: "whitespace and blanks are tolerated", csv: " ollama , , lmstudio ", want: []string{"ollama", "lmstudio"}}, {name: "duplicates collapse", csv: "ollama,ollama", want: []string{"ollama"}}, diff --git a/services/nvpair-ui-broker/main.go b/services/nvpair-ui-broker/main.go index 65e254cf..e5add56e 100644 --- a/services/nvpair-ui-broker/main.go +++ b/services/nvpair-ui-broker/main.go @@ -27,7 +27,7 @@ func main() { scannerPath := flag.String("scanner-path", "", "path to nvpair-node-scanner binary (default: ./nvpair-node-scanner in the current working directory)") nodeInfoPath := flag.String("node-info-path", "", "path to nvpair-node-info binary (default: ./nvpair-node-info in the current working directory)") proxyPath := flag.String("proxy-path", "", "path to nvpair-proxy binary (default: ./nvpair-proxy in the current working directory)") - proxyEngines := flag.String("proxy-engines", strings.Join(engines.Names(), ","), "comma-separated engines to front with a proxy; one nvpair-proxy process hosts a facade for each entry") + proxyEngines := flag.String("proxy-engines", strings.Join(defaultProxyEngineNames(), ","), "comma-separated engines to front with a proxy; one nvpair-proxy process hosts a facade for each entry") workloadMgrPath := flag.String("workload-manager-path", "", "path to nvpair-workload-manager binary (default: ./nvpair-workload-manager in the current working directory)") errorsPath := flag.String("errors-path", "", "path to nvpair-errors binary (default: ./nvpair-errors in the current working directory)") engineMgrPath := flag.String("engine-manager-path", "", "path to nvpair-engine-manager binary (default: ./nvpair-engine-manager in the current working directory)") @@ -310,6 +310,15 @@ func resolveProxyPath(override string) (string, error) { return resolveSiblingBinary(override, "nvpair-proxy", "--proxy-path") } +func defaultProxyEngineNames() []string { + defaults := engines.ProxyDefaults() + names := make([]string, len(defaults)) + for i, engine := range defaults { + names[i] = engine.Name + } + return names +} + // parseProxyEngines narrows the --proxy-engines list against the shared engine // table. An unknown name fails rather than being skipped: the operator asked // for an engine that does not exist, and quietly fronting the others would diff --git a/services/shared/engines/engines.go b/services/shared/engines/engines.go index b7f1e8c0..f3180677 100644 --- a/services/shared/engines/engines.go +++ b/services/shared/engines/engines.go @@ -104,6 +104,10 @@ type Engine struct { // broker reads it when reserving ports away from the OLLAMA_HOST alias, // so the two must agree — which is why it lives here. PortFile string + + // ProxyEnabledByDefault controls whether broker and TUI startup includes + // this engine's facade without an explicit --proxy-engines selection. + ProxyEnabledByDefault bool } // ProxyComponent is the proxy *process* identity, as distinct from the @@ -128,20 +132,22 @@ func (e Engine) ComponentName() string { return e.Name + "-proxy" } // all is the ordered engine set. Ollama is first; see the package comment. var all = []Engine{ { - Name: "ollama", - DisplayName: "Ollama", - DiscoveryService: noderec.ServiceOllama, - FacadePort: 11434, - EnginePortBase: 11435, - PortFile: "proxy-port.json", + Name: "ollama", + DisplayName: "Ollama", + DiscoveryService: noderec.ServiceOllama, + FacadePort: 11434, + EnginePortBase: 11435, + PortFile: "proxy-port.json", + ProxyEnabledByDefault: true, }, { - Name: "lmstudio", - DisplayName: "LM Studio", - DiscoveryService: noderec.ServiceLMStudio, - FacadePort: 1234, - EnginePortBase: 1235, - PortFile: "lmstudio-proxy-port.json", + Name: "lmstudio", + DisplayName: "LM Studio", + DiscoveryService: noderec.ServiceLMStudio, + FacadePort: 1234, + EnginePortBase: 1235, + PortFile: "lmstudio-proxy-port.json", + ProxyEnabledByDefault: true, }, } @@ -153,6 +159,18 @@ func All() []Engine { return out } +// ProxyDefaults returns the engines whose facades start without an explicit +// selection, preserving the shared preparation order. +func ProxyDefaults() []Engine { + out := make([]Engine, 0, len(all)) + for _, e := range all { + if e.ProxyEnabledByDefault { + out = append(out, e) + } + } + return out +} + // Names returns every engine id in preparation order. func Names() []string { out := make([]string, len(all)) diff --git a/services/shared/engines/engines_test.go b/services/shared/engines/engines_test.go index aa609be8..263e1393 100644 --- a/services/shared/engines/engines_test.go +++ b/services/shared/engines/engines_test.go @@ -34,6 +34,19 @@ func TestNames(t *testing.T) { } } +func TestProxyDefaults(t *testing.T) { + got := ProxyDefaults() + want := []string{"ollama", "lmstudio"} + if len(got) != len(want) { + t.Fatalf("ProxyDefaults() = %v, want %v", got, want) + } + for i, name := range want { + if got[i].Name != name || !got[i].ProxyEnabledByDefault { + t.Fatalf("ProxyDefaults() = %v, want %v", got, want) + } + } +} + // There are two proxy identities: ComponentName per facade, ProxyComponent per // process. Ollama's relay prefix was once the bare "proxy" while its error IDs // were already "ollama-proxy" — two near-identical values that coincided for LM From 76011d2f019fa52f3da1905766f2e46cc13fe66b Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 01:17:19 -0700 Subject: [PATCH 05/70] refactor(broker): generalize facade status subscriptions Signed-off-by: Sherief Farouk --- services/nvpair-ui-broker/broker.go | 101 +++--------------- services/nvpair-ui-broker/engineproxy.go | 46 ++++++++ .../nvpair-ui-broker/proxyaddressing_test.go | 83 ++++++++++++++ 3 files changed, 142 insertions(+), 88 deletions(-) diff --git a/services/nvpair-ui-broker/broker.go b/services/nvpair-ui-broker/broker.go index c5da7fd1..38416401 100644 --- a/services/nvpair-ui-broker/broker.go +++ b/services/nvpair-ui-broker/broker.go @@ -133,11 +133,11 @@ type SubscriptionResult struct { Subscribed bool `json:"subscribed"` } -// ProxyStatusResult is the response to "ollama-proxy:get-status". Ready is false -// (and Port 0) until the supervised ollama-proxy has emitted its "ready" -// notification — or always, if no proxy is being supervised. Clients poll -// this to learn where the local proxy is listening, since the proxy is -// optional and comes up asynchronously after app:ready. +// ProxyStatusResult is the response to "-proxy:get-status". Ready is +// false (and Port 0) until that facade has emitted its "ready" notification — +// or always, if no proxy is being supervised. Clients poll this to learn where +// the local facade is listening, since the proxy is optional and comes up +// asynchronously after app:ready. type ProxyStatusResult struct { Ready bool `json:"ready"` Port int `json:"port"` @@ -3142,6 +3142,13 @@ func (b *Broker) handleMessage(msg *Message) { return } + if profile, ok := engineProxyProfileForMethod(msg.Method); ok { + method := strings.TrimPrefix(msg.Method, profile.ComponentName()+":") + if b.handleEngineProxyBrokerRequest(profile, method, msg) { + return + } + } + switch msg.Method { case "engine:get-settings", "engine:preview-settings", "engine:apply-settings": go b.handleEngineSettings(msg) @@ -3189,50 +3196,6 @@ func (b *Broker) handleMessage(msg *Message) { log.Printf("failed to respond to discovery:unsubscribe: %v", err) } - case "ollama-proxy:get-status": - // Answered locally from the proxy handle's captured state — no - // round-trip to the proxy. When no proxy is supervised the - // zero value ({ready:false, port:0}) is a valid "not available" - // answer, so the method never errors. - var result ProxyStatusResult - if p := b.getProxy(); p != nil { - ready, port := p.Status(ollamaProxyProfile.Name) - result.Ready = ready - result.Port = port - } - if err := b.codec.Respond(msg.ID, result); err != nil { - log.Printf("failed to respond to proxy:get-status: %v", err) - } - - case "ollama-proxy:subscribe": - b.proxyMu.Lock() - wasSubscribed := b.setEngineProxySubscribed(ollamaProxyProfile, true) - b.proxyMu.Unlock() - if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: true}); err != nil { - log.Printf("failed to respond to proxy:subscribe: %v", err) - } - // On a fresh subscription replay the proxy's last "ready" payload - // (if it has come up) as a baseline proxy:ready, so a subscriber - // learns the port without a separate proxy:get-status. Sent after - // the ack. A redundant re-subscribe doesn't re-emit. - if !wasSubscribed { - if p := b.getProxy(); p != nil { - if rp := p.ReadyParams(ollamaProxyProfile.Name); rp != nil { - if err := b.codec.Notify("ollama-proxy:ready", rp); err != nil { - slog.Warn("emit baseline proxy:ready failed", "err", err) - } - } - } - } - - case "ollama-proxy:unsubscribe": - b.proxyMu.Lock() - b.setEngineProxySubscribed(ollamaProxyProfile, false) - b.proxyMu.Unlock() - if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: false}); err != nil { - log.Printf("failed to respond to proxy:unsubscribe: %v", err) - } - case "engine:set-port": go b.handleSettingsPortRPC(msg, "") case "ollama-proxy:set-port": @@ -3240,44 +3203,6 @@ func (b *Broker) handleMessage(msg *Message) { case "lmstudio-proxy:set-port": go b.handleSettingsPortRPC(msg, "lmstudio") - case "lmstudio-proxy:get-status": - // Answered locally from the lmstudio-proxy handle's captured state, - // mirroring proxy:get-status. Zero value when none is supervised. - var result ProxyStatusResult - if p := b.getLMStudioProxy(); p != nil { - ready, port := p.Status(lmstudioProxyProfile.Name) - result.Ready = ready - result.Port = port - } - if err := b.codec.Respond(msg.ID, result); err != nil { - log.Printf("failed to respond to lmstudio-proxy:get-status: %v", err) - } - - case "lmstudio-proxy:subscribe": - b.proxyMu.Lock() - wasSubscribed := b.setEngineProxySubscribed(lmstudioProxyProfile, true) - b.proxyMu.Unlock() - if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: true}); err != nil { - log.Printf("failed to respond to lmstudio-proxy:subscribe: %v", err) - } - if !wasSubscribed { - if p := b.getLMStudioProxy(); p != nil { - if rp := p.ReadyParams(lmstudioProxyProfile.Name); rp != nil { - if err := b.codec.Notify("lmstudio-proxy:ready", rp); err != nil { - slog.Warn("emit baseline lmstudio-proxy:ready failed", "err", err) - } - } - } - } - - case "lmstudio-proxy:unsubscribe": - b.proxyMu.Lock() - b.setEngineProxySubscribed(lmstudioProxyProfile, false) - b.proxyMu.Unlock() - if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: false}); err != nil { - log.Printf("failed to respond to lmstudio-proxy:unsubscribe: %v", err) - } - case "workloads:subscribe": b.workloadsMu.Lock() b.workloadsSubscribed = true @@ -3365,7 +3290,7 @@ func (b *Broker) handleMessage(msg *Message) { // Any remaining method under an engine's : prefix is // relayed verbatim to that engine's proxy (the reserved broker-local // ones — get-status and the subscription methods — are handled by - // their own cases above). This makes the broker a thin pass-through + // the profile-driven block above). This makes the broker a thin pass-through // for each proxy's whole control plane without enumerating methods. // // The prefixes are the engines' ComponentName values, so this loop diff --git a/services/nvpair-ui-broker/engineproxy.go b/services/nvpair-ui-broker/engineproxy.go index 1254d08e..5d82d679 100644 --- a/services/nvpair-ui-broker/engineproxy.go +++ b/services/nvpair-ui-broker/engineproxy.go @@ -509,6 +509,52 @@ func (b *Broker) setEngineProxySubscribed(p engineProxyProfile, subscribed bool) return was } +// handleEngineProxyBrokerRequest serves the facade methods owned by the broker +// rather than the proxy child. It reports whether method was handled. +func (b *Broker) handleEngineProxyBrokerRequest(profile engineProxyProfile, method string, msg *Message) bool { + switch method { + case "get-status": + var result ProxyStatusResult + if proxy := b.engineProxyHandle(profile); proxy != nil { + result.Ready, result.Port = proxy.Status(profile.Name) + } + if err := b.codec.Respond(msg.ID, result); err != nil { + log.Printf("failed to respond to %s:get-status: %v", profile.ComponentName(), err) + } + return true + + case "subscribe": + b.proxyMu.Lock() + wasSubscribed := b.setEngineProxySubscribed(profile, true) + b.proxyMu.Unlock() + if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: true}); err != nil { + log.Printf("failed to respond to %s:subscribe: %v", profile.ComponentName(), err) + } + // The acknowledgement must precede the baseline notification. A + // redundant subscription is already live and needs no replay. + if !wasSubscribed { + if proxy := b.engineProxyHandle(profile); proxy != nil { + if params := proxy.ReadyParams(profile.Name); params != nil { + if err := b.codec.Notify(profile.ComponentName()+":ready", params); err != nil { + slog.Warn("emit baseline proxy ready failed", "engine", profile.Name, "err", err) + } + } + } + } + return true + + case "unsubscribe": + b.proxyMu.Lock() + b.setEngineProxySubscribed(profile, false) + b.proxyMu.Unlock() + if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: false}); err != nil { + log.Printf("failed to respond to %s:unsubscribe: %v", profile.ComponentName(), err) + } + return true + } + return false +} + // relayToEngineProxy forwards an -proxy: client request to that // engine's facade and maps the response straight back. // diff --git a/services/nvpair-ui-broker/proxyaddressing_test.go b/services/nvpair-ui-broker/proxyaddressing_test.go index 533b1769..9b5192b8 100644 --- a/services/nvpair-ui-broker/proxyaddressing_test.go +++ b/services/nvpair-ui-broker/proxyaddressing_test.go @@ -234,6 +234,89 @@ func TestReadinessIsTrackedPerEngine(t *testing.T) { } } +func TestBrokerOwnedFacadeMethodsFollowTheProfile(t *testing.T) { + for _, profile := range engineProxyProfiles { + t.Run(profile.Name, func(t *testing.T) { + client, server := net.Pipe() + t.Cleanup(func() { + _ = client.Close() + _ = server.Close() + }) + payload := json.RawMessage(fmt.Sprintf(`{"version":"test","port":%d}`, profile.FacadePort)) + proxy := &proxyProcess{facadeState: map[string]proxyFacadeState{ + profile.Name: { + ready: true, + port: profile.FacadePort, + params: payload, + }, + }} + b := &Broker{codec: NewCodec(server)} + b.setEngineProxyHandle(profile, proxy) + reader := NewCodec(client) + nextID := 0 + call := func(method string, frameCount int) []*Message { + t.Helper() + nextID++ + id := json.RawMessage(fmt.Sprintf("%d", nextID)) + done := make(chan struct{}) + go func() { + b.handleMessage(&Message{JSONRPC: "2.0", ID: &id, Method: profile.ComponentName() + ":" + method}) + close(done) + }() + frames := make([]*Message, 0, frameCount) + for range frameCount { + if err := client.SetReadDeadline(time.Now().Add(2 * time.Second)); err != nil { + t.Fatalf("set read deadline: %v", err) + } + frame, err := reader.Read() + if err != nil { + t.Fatalf("read %s frame: %v", method, err) + } + frames = append(frames, frame) + } + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatalf("%s handler did not finish after %d frame(s)", method, frameCount) + } + return frames + } + + statusFrames := call("get-status", 1) + var status ProxyStatusResult + if err := json.Unmarshal(statusFrames[0].Result, &status); err != nil { + t.Fatalf("decode status: %v", err) + } + if !status.Ready || status.Port != profile.FacadePort { + t.Fatalf("status = %+v, want ready on %d", status, profile.FacadePort) + } + + subscribeFrames := call("subscribe", 2) + var subscribed SubscriptionResult + if err := json.Unmarshal(subscribeFrames[0].Result, &subscribed); err != nil || !subscribed.Subscribed { + t.Fatalf("subscribe response = %s, error %v", subscribeFrames[0].Result, err) + } + if subscribeFrames[1].Method != profile.ComponentName()+":ready" { + t.Fatalf("second subscribe frame = %q, want ready baseline after response", subscribeFrames[1].Method) + } + if string(subscribeFrames[1].Params) != string(payload) { + t.Fatalf("ready baseline = %s, want %s", subscribeFrames[1].Params, payload) + } + + unsubscribeFrames := call("unsubscribe", 1) + if err := json.Unmarshal(unsubscribeFrames[0].Result, &subscribed); err != nil || subscribed.Subscribed { + t.Fatalf("unsubscribe response = %s, error %v", unsubscribeFrames[0].Result, err) + } + b.proxyMu.Lock() + stillSubscribed := b.engineProxySubscribed(profile) + b.proxyMu.Unlock() + if stillSubscribed { + t.Fatal("facade remained subscribed after unsubscribe") + } + }) + } +} + // Each facade subscribes for its own engine's discovery service, so the broker // has to track a subscription per engine. A single id per process let the second // facade's subscribe replace the first's, which unsubscribed a live facade and From ffa167a9d68f8e67247e22cf7840228fee932c29 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 01:28:15 -0700 Subject: [PATCH 06/70] feat(broker): support prepositioned process facades Signed-off-by: Sherief Farouk --- services/nvpair-ui-broker/broker.go | 32 +++- services/nvpair-ui-broker/engineproxy.go | 91 +++++++++--- .../prepositionedproxy_test.go | 140 ++++++++++++++++++ services/nvpair-ui-broker/proxyport.go | 3 + 4 files changed, 246 insertions(+), 20 deletions(-) create mode 100644 services/nvpair-ui-broker/prepositionedproxy_test.go diff --git a/services/nvpair-ui-broker/broker.go b/services/nvpair-ui-broker/broker.go index 38416401..5a7c330f 100644 --- a/services/nvpair-ui-broker/broker.go +++ b/services/nvpair-ui-broker/broker.go @@ -755,6 +755,26 @@ func (b *Broker) proxyBringUpContext() (context.Context, context.CancelFunc) { func (b *Broker) enableEngineFacade( ctx context.Context, pp *proxyProcess, profile engineProxyProfile, alias ollamaHostAlias, ) error { + return b.enableEngineFacadeWithPortCheck(ctx, pp, profile, alias, tcpPortAvailable) +} + +func (b *Broker) enableEngineFacadeWithPortCheck( + ctx context.Context, + pp *proxyProcess, + profile engineProxyProfile, + alias ollamaHostAlias, + available func(int) bool, +) error { + if profile.Ownership == prepositionedEngine { + return b.enableProxyFacadeWithFallback( + ctx, + pp, + b.prepositionedFacadeSpec(profile), + func(failed int) int { + return b.prepositionedFallbackPortWithCheck(profile, failed, available) + }, + ) + } switch profile.Name { case ollamaProxyProfile.Name: return b.enableProxyFacadeWithFallback(ctx, pp, b.ollamaFacadeSpec(alias), b.ollamaFallbackPort) @@ -775,6 +795,9 @@ func (b *Broker) enableEngineFacade( // alias. Finishing is what returns the alias, which is only correct once this // engine is known not to be coming up. func (b *Broker) blockAndFinishEngineProxy(profile engineProxyProfile) { + if profile.Ownership == prepositionedEngine { + return + } switch profile.Name { case ollamaProxyProfile.Name: if b.ollamaState().managedFacade.Load() { @@ -981,8 +1004,13 @@ func (b *Broker) forwardProxyProcessNotification( slog.Debug("ignoring unaddressed proxy notification", "method", bare) } default: - slog.Warn("proxy addressed a notification to an unknown engine", - "engine", engine, "method", bare) + profile, known := engineProxyProfileFor(engine) + if known && profile.Ownership == prepositionedEngine { + b.forwardPrepositionedProxyNotification(profile, method, params) + return + } + slog.Warn("proxy addressed a notification without a handler", + "engine", engine, "method", bare, "known", known) } } diff --git a/services/nvpair-ui-broker/engineproxy.go b/services/nvpair-ui-broker/engineproxy.go index 5d82d679..d37823de 100644 --- a/services/nvpair-ui-broker/engineproxy.go +++ b/services/nvpair-ui-broker/engineproxy.go @@ -30,8 +30,7 @@ import ( "nvpair-shared/noderec" ) -// engineOwnership answers a single question: may the broker reposition this -// engine's process while it is running? +// engineOwnership selects the broker's port-ownership strategy. type engineOwnership int const ( @@ -45,6 +44,11 @@ const ( // and has an official stop command for it, so it may be stopped and // repositioned before the proxy starts. LM Studio. managedEngine + + // prepositionedEngine — engine-manager already owns the process and keeps + // it on EnginePortBase. The broker only places the facade, never gates + // engine requests or takes over/repositions the backend. + prepositionedEngine ) // engineProxyProfile is everything the broker needs to supervise one engine's @@ -52,9 +56,9 @@ const ( type engineProxyProfile struct { engines.Engine - // Ownership decides the port choreography: plan-then-commit for an - // adopted engine, move-then-verify for a managed one. It is the only - // judgment call in adding an engine. + // Ownership decides the port choreography: plan-then-commit for an adopted + // engine, move-then-verify for a managed one, or facade-only placement for + // a prepositioned one. It is the only judgment call in adding an engine. Ownership engineOwnership // HealthProbePath is the path whose 200 means "this engine is answering". @@ -295,6 +299,34 @@ func (b *Broker) enableProxyFacadeWithFallback( return b.enableProxyFacade(parent, p, spec) } +// prepositionedFacadeSpec prefers the stock facade port, while preserving a +// fallback or explicit port already selected for this broker lifetime. +func (b *Broker) prepositionedFacadeSpec(profile engineProxyProfile) enableFacadeRequest { + spec := enableFacadeRequest{Engine: profile.Name, Port: profile.FacadePort} + if port := int(b.engineProxy(profile).startupPort.Load()); port != 0 { + spec.Port = port + spec.IgnorePersistedPort = true + } + return spec +} + +// prepositionedFallbackPortWithCheck keeps a fallback off the fixed backend +// port and every sibling's facade/backend/persisted ports. +func (b *Broker) prepositionedFallbackPortWithCheck( + profile engineProxyProfile, failed int, available func(int) bool, +) int { + excluded := []int{failed, profile.EnginePortBase} + if alias := b.currentOllamaHostAlias().Port; alias > 0 { + excluded = append(excluded, alias) + } + for port := range b.siblingEngineProxyPorts(profile) { + excluded = append(excluded, port) + } + fallback := nextAvailablePortExcluding(profile.FacadePort, excluded, available) + b.engineProxy(profile).startupPort.Store(int32(fallback)) + return fallback +} + // facadeMethodFor strips a facade-scoped notification's engine address and // confirms it belongs to the engine this reader speaks for. // @@ -361,10 +393,10 @@ func (b *Broker) proxyEnabled(p engineProxyProfile) bool { return false } -// prepareEnabledFacades prepares managed port ownership for the engines the -// broker is actually going to front, in the table's order — Ollama first, -// because its preparation reserves any inherited OLLAMA_HOST alias that later -// engines must route around. +// prepareEnabledFacades prepares port ownership for the engines the broker is +// actually going to front, in table order — Ollama first, because its alias +// reservation constrains later engines. A prepositioned backend needs no gate +// or move; recording its fixed backend port is its entire preparation. // // The enablement check belongs here and not downstream, because preparation is // not read-only: for a managed engine the backend move runs inside it, so @@ -373,14 +405,27 @@ func (b *Broker) proxyEnabled(p engineProxyProfile) bool { // symptom, since its move is deferred until its proxy proves it holds the // facade — which is exactly why this cannot be left to the callee. func (b *Broker) prepareEnabledFacades() { - if b.proxyEnabled(ollamaProxyProfile) { - b.prepareManagedOllamaFacade() - } - if b.proxyEnabled(lmstudioProxyProfile) { - b.prepareManagedLMStudioFacade() + for _, profile := range engineProxyProfiles { + if !b.proxyEnabled(profile) { + continue + } + switch { + case profile.Name == ollamaProxyProfile.Name: + b.prepareManagedOllamaFacade() + case profile.Name == lmstudioProxyProfile.Name: + b.prepareManagedLMStudioFacade() + case profile.Ownership == prepositionedEngine: + b.preparePrepositionedFacade(profile) + default: + slog.Warn("no port preparation strategy for engine", "engine", profile.Name) + } } } +func (b *Broker) preparePrepositionedFacade(profile engineProxyProfile) { + b.engineProxy(profile).backendPort.Store(int32(profile.EnginePortBase)) +} + // proxyDisabledReason explains why an engine has no proxy, and reports whether // that is worth surfacing to the user rather than only logging it. // @@ -452,10 +497,6 @@ func (b *Broker) setEngineProxyHandle(p engineProxyProfile, proxy *proxyProcess) // // This is the end of the line for a notification — every path consumes it, so // there is nothing for a caller to do afterwards and nothing to report back. -// The heads of the two callers stay separate: the bind-failure and readiness -// handling genuinely differ by ownership, and folding them in behind a -// callback would move those bodies into this file without making them any more -// shared. func (b *Broker) forwardEngineProxyNotification(profile engineProxyProfile, method string, params json.RawMessage) { if b.routeProcessScopedProxyNotification(method, params) { return @@ -472,6 +513,20 @@ func (b *Broker) forwardEngineProxyNotification(profile engineProxyProfile, meth } } +// forwardPrepositionedProxyNotification is the facade-only notification path: +// there is no ownership or readiness reconciliation because the backend stays +// fixed on EnginePortBase. +func (b *Broker) forwardPrepositionedProxyNotification(profile engineProxyProfile, method string, params json.RawMessage) { + method, addressed := facadeMethodFor(profile, method) + if !addressed { + return + } + if b.dispatchErrorsNotif(profile.ComponentName(), method, params) { + return + } + b.forwardEngineProxyNotification(profile, method, params) +} + // routeProcessScopedProxyNotification handles the notifications that belong to // the proxy process rather than to one of its facades, and reports whether it // took the method. diff --git a/services/nvpair-ui-broker/prepositionedproxy_test.go b/services/nvpair-ui-broker/prepositionedproxy_test.go new file mode 100644 index 00000000..46bf4ffe --- /dev/null +++ b/services/nvpair-ui-broker/prepositionedproxy_test.go @@ -0,0 +1,140 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "context" + "encoding/json" + "net" + "testing" + "time" + + "nvpair-shared/engines" +) + +func testPrepositionedProfile() engineProxyProfile { + return engineProxyProfile{ + Engine: engines.Engine{ + Name: "fixedtest", + DisplayName: "Fixed Test", + FacadePort: 1233, + EnginePortBase: 1234, + PortFile: "fixedtest-proxy-port.json", + }, + Ownership: prepositionedEngine, + HealthProbePath: "/health", + } +} + +func brokerWithPrepositionedProfile(profile engineProxyProfile) *Broker { + b := &Broker{} + b.engineProxiesOnce.Do(func() { + b.engineProxies = map[string]*engineProxyRuntime{ + profile.Name: {profile: profile}, + } + }) + return b +} + +func TestPrepositionedFacadeRetriesAwayFromReservedPorts(t *testing.T) { + isolateOllamaHostTestConfig(t) + profile := testPrepositionedProfile() + b := brokerWithPrepositionedProfile(profile) + + proxyClient, proxyServer := net.Pipe() + t.Cleanup(func() { + _ = proxyClient.Close() + _ = proxyServer.Close() + }) + proxy := &proxyProcess{peer: NewPeer(NewCodec(proxyClient))} + go proxy.peer.Serve(nil, nil) + attempts := make(chan enableFacadeRequest, 2) + serveFacadeEnable(t, proxyServer, map[int]bool{profile.FacadePort: true}, attempts) + + err := b.enableEngineFacadeWithPortCheck( + context.Background(), + proxy, + profile, + ollamaHostAlias{}, + func(int) bool { return true }, + ) + if err != nil { + t.Fatalf("enable prepositioned facade: %v", err) + } + first, second := <-attempts, <-attempts + if first.Port != profile.FacadePort { + t.Fatalf("first port = %d, want stock facade %d", first.Port, profile.FacadePort) + } + // 1234 is this backend and LM Studio's facade; 1235 is LM Studio's backend. + if second.Port != 1236 { + t.Fatalf("fallback port = %d, want 1236 after backend and sibling exclusions", second.Port) + } + if !second.IgnorePersistedPort { + t.Fatal("fallback retry could restore the port that just failed") + } + + restart := b.prepositionedFacadeSpec(profile) + if restart.Port != second.Port || !restart.IgnorePersistedPort { + t.Fatalf("restart spec = %+v, want explicit fallback port %d", restart, second.Port) + } +} + +func TestPrepositionedProfileNeverTakesBackendOwnership(t *testing.T) { + profile := testPrepositionedProfile() + if profile.mayMoveRunningEngine() { + t.Fatal("prepositioned strategy may move a running backend") + } + if profile.blocksOnOccupiedFacade() { + t.Fatal("prepositioned strategy blocks instead of moving only its facade") + } + + b := brokerWithPrepositionedProfile(profile) + b.preparePrepositionedFacade(profile) + state := b.engineProxy(profile) + if got := int(state.backendPort.Load()); got != profile.EnginePortBase { + t.Fatalf("backend port = %d, want fixed base %d", got, profile.EnginePortBase) + } + state.managedFacade.Store(true) + b.blockAndFinishEngineProxy(profile) + b.finishEngineProxyStartup(profile) + if !state.managedFacade.Load() { + t.Fatal("ungated terminal handling mutated ownership state") + } +} + +func TestPrepositionedNotificationPreservesFacadeAddress(t *testing.T) { + profile := lmstudioProxyProfile + profile.Ownership = prepositionedEngine + client, server := net.Pipe() + t.Cleanup(func() { + _ = client.Close() + _ = server.Close() + }) + b := &Broker{codec: NewCodec(server)} + b.proxyMu.Lock() + b.setEngineProxySubscribed(profile, true) + b.proxyMu.Unlock() + + payload := json.RawMessage(`{"port":1234}`) + done := make(chan struct{}) + go func() { + b.forwardPrepositionedProxyNotification(profile, profile.addressed("ready"), payload) + close(done) + }() + if err := client.SetReadDeadline(time.Now().Add(2 * time.Second)); err != nil { + t.Fatalf("set read deadline: %v", err) + } + msg, err := NewCodec(client).Read() + if err != nil { + t.Fatalf("read forwarded notification: %v", err) + } + if msg.Method != profile.ComponentName()+":ready" || string(msg.Params) != string(payload) { + t.Fatalf("forwarded notification = %s %s", msg.Method, msg.Params) + } + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("notification forwarding did not finish") + } +} diff --git a/services/nvpair-ui-broker/proxyport.go b/services/nvpair-ui-broker/proxyport.go index 7ecfb7b5..01edd5f1 100644 --- a/services/nvpair-ui-broker/proxyport.go +++ b/services/nvpair-ui-broker/proxyport.go @@ -177,6 +177,9 @@ func (b *Broker) configureProxySupervisorCallbacks(sup *supervisor) { // than looping over a single helper: its gate re-checks whether a backend move // is pending, so clearing that move state is what actually reopens the path. func (b *Broker) finishEngineProxyStartup(profile engineProxyProfile) { + if profile.Ownership == prepositionedEngine { + return + } switch profile.Name { case ollamaProxyProfile.Name: b.finishOllamaProxyTerminal() From cc5b4a8ff093eb5f06a3a5db4966dcaa9bddb917 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 01:44:08 -0700 Subject: [PATCH 07/70] feat(broker): advertise standard engine profiles Signed-off-by: Sherief Farouk --- services/nvpair-ui-broker/advertiser.go | 37 ++++++ services/nvpair-ui-broker/advertiser_test.go | 106 ++++++++++++++++++ services/nvpair-ui-broker/broker.go | 26 ++++- .../nvpair-ui-broker/broker_lifecycle_test.go | 7 +- 4 files changed, 167 insertions(+), 9 deletions(-) diff --git a/services/nvpair-ui-broker/advertiser.go b/services/nvpair-ui-broker/advertiser.go index d48097e7..dcb0aa0a 100644 --- a/services/nvpair-ui-broker/advertiser.go +++ b/services/nvpair-ui-broker/advertiser.go @@ -188,6 +188,43 @@ func (b *Broker) reconcileAdvertiseLMStudio(client *http.Client) { } } +// runAutoAdvertisePrepositioned reconciles an ungated engine whose backend is +// fixed on the port recorded in its runtime profile. +func (b *Broker) runAutoAdvertisePrepositioned(ctx context.Context, profile engineProxyProfile) { + client := &http.Client{Timeout: 2 * time.Second} + ticker := time.NewTicker(autoAdvertiseInterval) + defer ticker.Stop() + + b.reconcileAdvertisePrepositioned(profile, client) + for { + select { + case <-ctx.Done(): + return + case <-ticker.C: + b.reconcileAdvertisePrepositioned(profile, client) + } + } +} + +func (b *Broker) reconcileAdvertisePrepositioned(profile engineProxyProfile, client *http.Client) { + b.engineConfigMu.Lock() + defer b.engineConfigMu.Unlock() + + enginePort := int(b.engineProxy(profile).backendPort.Load()) + proxyPort := b.engineProxyListenPort(profile) + up := enginePort > 0 && + proxyPort > 0 && + enginePort != proxyPort && + checkEngineHealth(profile, client, enginePort) + if up { + b.registerService(noderec.RegisterParams{Service: profile.DiscoveryService, Port: proxyPort}) + b.setProxyLocalBackend(b.engineProxyHandle(profile), profile.Name, enginePort, true) + return + } + b.unregisterService(profile.DiscoveryService) + b.setProxyLocalBackend(b.engineProxyHandle(profile), profile.Name, enginePort, false) +} + // proxyLocalBackend is the node/set-local-backend payload: the loopback engine // the proxy's cluster mTLS ingress forwards to, and the proxy's own self // candidate on the local routing path. diff --git a/services/nvpair-ui-broker/advertiser_test.go b/services/nvpair-ui-broker/advertiser_test.go index a32c2cbe..8b42a06c 100644 --- a/services/nvpair-ui-broker/advertiser_test.go +++ b/services/nvpair-ui-broker/advertiser_test.go @@ -6,9 +6,13 @@ package main import ( "encoding/json" "net" + "net/http" + "net/http/httptest" + "strconv" "testing" "time" + "nvpair-shared/noderec" "nvpair-ui-broker/relay" ) @@ -107,6 +111,108 @@ func TestLMStudioFallbackDoesNotOverwriteKnownBackend(t *testing.T) { } } +func TestPrepositionedAdvertiserTracksBackendHealth(t *testing.T) { + backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusOK) + })) + defer backend.Close() + _, portText, err := net.SplitHostPort(backend.Listener.Addr().String()) + if err != nil { + t.Fatal(err) + } + backendPort, err := strconv.Atoi(portText) + if err != nil { + t.Fatal(err) + } + proxyPort := 44000 + if backendPort == proxyPort { + proxyPort++ + } + + profile := testPrepositionedProfile() + profile.DiscoveryService = noderec.ServiceLMStudio + b := brokerWithPrepositionedProfile(profile) + b.regCache = relay.NewRegistrationCache() + b.engineProxy(profile).backendPort.Store(int32(backendPort)) + updates := attachAdvertiserProxy(t, b, profile, proxyPort) + + b.reconcileAdvertisePrepositioned(profile, backend.Client()) + registrations := b.regCache.Snapshot() + if len(registrations) != 1 || registrations[0].Service != profile.DiscoveryService || + registrations[0].Port != proxyPort { + t.Fatalf("healthy registration = %+v, want %s on %d", registrations, profile.DiscoveryService, proxyPort) + } + if got := <-updates; !got.Healthy || got.Port != backendPort || got.Engine != profile.Name { + t.Fatalf("healthy local backend = %+v", got) + } + + backend.Close() + b.reconcileAdvertisePrepositioned(profile, backend.Client()) + if got := b.regCache.Snapshot(); len(got) != 0 { + t.Fatalf("unhealthy engine remained advertised: %+v", got) + } + if got := <-updates; got.Healthy || got.Port != backendPort { + t.Fatalf("unhealthy local backend = %+v", got) + } +} + +func TestPrepositionedAdvertiserRejectsSelfForwardLoop(t *testing.T) { + profile := testPrepositionedProfile() + profile.DiscoveryService = noderec.ServiceLMStudio + b := brokerWithPrepositionedProfile(profile) + b.regCache = relay.NewRegistrationCache() + b.regCache.Register(noderec.RegisterParams{Service: profile.DiscoveryService, Port: 44000}) + b.engineProxy(profile).backendPort.Store(44000) + updates := attachAdvertiserProxy(t, b, profile, 44000) + + // A nil client proves the collision check short-circuits before probing the + // facade as though it were the backend. + b.reconcileAdvertisePrepositioned(profile, nil) + if got := b.regCache.Snapshot(); len(got) != 0 { + t.Fatalf("self-forwarding facade remained advertised: %+v", got) + } + if got := <-updates; got.Healthy { + t.Fatalf("self-forwarding backend remained healthy: %+v", got) + } +} + +func attachAdvertiserProxy( + t *testing.T, + b *Broker, + profile engineProxyProfile, + port int, +) <-chan proxyLocalBackend { + t.Helper() + proxyClient, proxyServer := net.Pipe() + t.Cleanup(func() { + _ = proxyClient.Close() + _ = proxyServer.Close() + }) + proxy := &proxyProcess{ + peer: NewPeer(NewCodec(proxyClient)), + facadeState: readyFacade(profile.Name, port), + } + go proxy.peer.Serve(nil, nil) + b.setEngineProxyHandle(profile, proxy) + + updates := make(chan proxyLocalBackend, 4) + go func() { + codec := NewCodec(proxyServer) + for { + msg, err := codec.Read() + if err != nil { + return + } + var update proxyLocalBackend + if json.Unmarshal(msg.Params, &update) == nil { + updates <- update + } + _ = codec.Respond(msg.ID, map[string]bool{"ok": true}) + } + }() + return updates +} + // TestProxyListenPortNoProxy: with no proxy supervised, proxyListenPort is 0, // so the self-forward collision check (port == proxy port) never falsely trips. func TestProxyListenPortNoProxy(t *testing.T) { diff --git a/services/nvpair-ui-broker/broker.go b/services/nvpair-ui-broker/broker.go index 5a7c330f..e83a9d25 100644 --- a/services/nvpair-ui-broker/broker.go +++ b/services/nvpair-ui-broker/broker.go @@ -537,14 +537,18 @@ func (b *Broker) restoreEnabledEnginesAfterPortGate(ctx context.Context) bool { func (b *Broker) runEngineAvailabilityAfterPortGates( ctx context.Context, - runOllama func(context.Context), - runLMStudio func(context.Context), + runners ...func(context.Context), ) bool { if !b.restoreEnabledEnginesAfterPortGate(ctx) { return false } - go runOllama(ctx) - runLMStudio(ctx) + for index, run := range runners { + if index == len(runners)-1 { + run(ctx) + break + } + go run(ctx) + } return true } @@ -2224,11 +2228,21 @@ func (b *Broker) Serve(ctx context.Context) error { } } - // Restore engines and begin both advertising loops only after both proxy + // Restore engines and begin advertising only after both managed proxy // startup attempts have established either readiness or a terminal outcome. // This prevents a restored engine from taking a persisted proxy port before // the broker can resolve ownership. - go b.runEngineAvailabilityAfterPortGates(ctx, b.runAutoAdvertise, b.runAutoAdvertiseLMStudio) + availabilityRunners := []func(context.Context){b.runAutoAdvertise, b.runAutoAdvertiseLMStudio} + for _, profile := range engineProxyProfiles { + if profile.Ownership != prepositionedEngine || !b.proxyEnabled(profile) { + continue + } + profile := profile + availabilityRunners = append(availabilityRunners, func(ctx context.Context) { + b.runAutoAdvertisePrepositioned(ctx, profile) + }) + } + go b.runEngineAvailabilityAfterPortGates(ctx, availabilityRunners...) // nvpair-workload-manager is another auxiliary worker: it relays local // workload lifecycle events to peer nodes and surfaces peer events diff --git a/services/nvpair-ui-broker/broker_lifecycle_test.go b/services/nvpair-ui-broker/broker_lifecycle_test.go index f60fcbe5..ef17b83b 100644 --- a/services/nvpair-ui-broker/broker_lifecycle_test.go +++ b/services/nvpair-ui-broker/broker_lifecycle_test.go @@ -54,7 +54,7 @@ func TestEngineAvailabilityWaitsForBothProxyOutcomes(t *testing.T) { restore <- msg.Method } }() - advertised := make(chan string, 2) + advertised := make(chan string, 3) ctx, cancel := context.WithCancel(context.Background()) defer cancel() done := make(chan bool, 1) @@ -63,6 +63,7 @@ func TestEngineAvailabilityWaitsForBothProxyOutcomes(t *testing.T) { ctx, func(context.Context) { advertised <- "ollama" }, func(context.Context) { advertised <- "lmstudio" }, + func(context.Context) { advertised <- "llamacpp" }, ) }() @@ -92,12 +93,12 @@ func TestEngineAvailabilityWaitsForBothProxyOutcomes(t *testing.T) { t.Fatal("enabled-engine restore did not run after both proxy outcomes") } seen := map[string]bool{} - for len(seen) < 2 { + for len(seen) < 3 { select { case got := <-advertised: seen[got] = true case <-time.After(2 * time.Second): - t.Fatalf("advertising did not start for both engines: %v", seen) + t.Fatalf("advertising did not start for every engine: %v", seen) } } if !<-done { From 92dda1dd8562a5f62453ed54dcb36a44998985d3 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 01:59:40 -0700 Subject: [PATCH 08/70] feat(engine-manager): support verified multi-artifact installs Signed-off-by: Sherief Farouk --- services/nvpair-engine-manager/install.go | 66 +++++++- .../install_artifacts_test.go | 157 ++++++++++++++++++ services/nvpair-engine-manager/registry.go | 73 ++++++-- .../nvpair-engine-manager/registry_test.go | 54 ++++++ 4 files changed, 334 insertions(+), 16 deletions(-) create mode 100644 services/nvpair-engine-manager/install_artifacts_test.go diff --git a/services/nvpair-engine-manager/install.go b/services/nvpair-engine-manager/install.go index 268bd0b9..302788b4 100644 --- a/services/nvpair-engine-manager/install.go +++ b/services/nvpair-engine-manager/install.go @@ -96,6 +96,18 @@ func (e *Executor) Install(ctx context.Context, engine string) error { vars["download"] = dp e.emitInstallProgress(engine, "verified", 50) } + if len(inst.Artifacts) > 0 { + paths, artifactVars, err := e.downloadInstallArtifacts(ctx, engine, inst.Artifacts) + defer removeDownloadedFiles(paths) + if err != nil { + e.reportInstallFailed(engine, err) + return err + } + for name, downloadPath := range artifactVars { + vars[name] = downloadPath + } + e.emitInstallProgress(engine, "verified", 50) + } if len(inst.Run) > 0 { e.emitInstallProgress(engine, "installing", 75) args, err := resolveArgs(inst.Run, vars) @@ -234,6 +246,9 @@ func validateDownloadURL(raw string) error { if err != nil { return fmt.Errorf("invalid download url %q: %w", raw, err) } + if u.Hostname() == "" { + return fmt.Errorf("download url %q must include a host", raw) + } switch u.Scheme { case "https": return nil @@ -248,6 +263,53 @@ func validateDownloadURL(raw string) error { } func (e *Executor) download(ctx context.Context, engine string, f *Fetch) (string, error) { + return e.downloadWithProgress(ctx, engine, f, func(percent int) { + e.emitInstallProgress(engine, "downloading", percent) + }) +} + +func (e *Executor) downloadInstallArtifacts( + ctx context.Context, + engine string, + artifacts []InstallArtifact, +) ([]string, map[string]string, error) { + paths := make([]string, 0, len(artifacts)) + vars := make(map[string]string, len(artifacts)) + e.emitInstallProgress(engine, "downloading", 0) + for index, artifact := range artifacts { + fetch := &Fetch{URL: artifact.URL, SHA256: artifact.SHA256} + downloadPath, err := e.downloadWithProgress(ctx, engine, fetch, func(percent int) { + e.emitInstallProgress(engine, "downloading", aggregateArtifactProgress(index, len(artifacts), percent)) + }) + if err != nil { + return paths, vars, fmt.Errorf("download artifact %q: %w", artifact.Name, err) + } + paths = append(paths, downloadPath) + vars["download_"+artifact.Name] = downloadPath + e.emitInstallProgress(engine, "downloading", aggregateArtifactProgress(index, len(artifacts), 100)) + } + return paths, vars, nil +} + +func aggregateArtifactProgress(index, count, percent int) int { + percent = max(0, min(percent, 100)) + start := index * 50 / count + end := (index + 1) * 50 / count + return start + (end-start)*percent/100 +} + +func removeDownloadedFiles(paths []string) { + for _, downloadPath := range paths { + _ = os.Remove(downloadPath) + } +} + +func (e *Executor) downloadWithProgress( + ctx context.Context, + engine string, + f *Fetch, + onProgress func(int), +) (string, error) { if err := validateDownloadURL(f.URL); err != nil { return "", err } @@ -273,9 +335,7 @@ func (e *Executor) download(ctx context.Context, engine string, f *Fetch) (strin if err != nil { return "", err } - pw := &progressWriter{total: resp.ContentLength, onPct: func(p int) { - e.emitInstallProgress(engine, "downloading", p) - }} + pw := &progressWriter{total: resp.ContentLength, onPct: onProgress} h := sha256.New() // Read one byte past the cap so we can detect (and reject) overflow. n, err := io.Copy(io.MultiWriter(tmp, h), io.TeeReader(io.LimitReader(resp.Body, maxDownloadBytes+1), pw)) diff --git a/services/nvpair-engine-manager/install_artifacts_test.go b/services/nvpair-engine-manager/install_artifacts_test.go new file mode 100644 index 00000000..39e577b5 --- /dev/null +++ b/services/nvpair-engine-manager/install_artifacts_test.go @@ -0,0 +1,157 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +func TestInstallDownloadsAllNamedArtifactsBeforeRunning(t *testing.T) { + payload := []byte("artifact") + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + _, _ = w.Write(payload) + })) + defer server.Close() + + marker := filepath.Join(t.TempDir(), "installed.json") + artifacts := []InstallArtifact{ + pinnedArtifact("server", server.URL+"/server.zip", payload), + pinnedArtifact("cudart", server.URL+"/cudart.zip", payload), + } + manifest := artifactInstallManifest(t, "artifact-success", marker, artifacts) + executor := newTestExecutor(t, manifest) + + if err := executor.Install(context.Background(), manifest.Engine); err != nil { + t.Fatalf("install named artifacts: %v", err) + } + data, err := os.ReadFile(marker) + if err != nil { + t.Fatalf("read captured install arguments: %v", err) + } + var downloads []string + if err := json.Unmarshal(data, &downloads); err != nil { + t.Fatalf("decode captured install arguments: %v", err) + } + if len(downloads) != 2 || downloads[0] == downloads[1] { + t.Fatalf("resolved artifact paths = %v", downloads) + } + assertNoArtifactTemps(t, manifest.Engine) +} + +func TestInstallRejectsBadSecondArtifactBeforeCommand(t *testing.T) { + payload := []byte("artifact") + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + _, _ = w.Write(payload) + })) + defer server.Close() + + marker := filepath.Join(t.TempDir(), "must-not-exist") + artifacts := []InstallArtifact{ + pinnedArtifact("server", server.URL+"/server.zip", payload), + {Name: "cudart", URL: server.URL + "/cudart.zip", SHA256: strings.Repeat("0", 64)}, + } + manifest := artifactInstallManifest(t, "artifact-bad-checksum", marker, artifacts) + executor := newTestExecutor(t, manifest) + + err := executor.Install(context.Background(), manifest.Engine) + if err == nil || !strings.Contains(err.Error(), "checksum mismatch") { + t.Fatalf("install error = %v, want checksum mismatch", err) + } + if fileExists(marker) { + t.Fatal("install command ran after an artifact checksum failed") + } + assertNoArtifactTemps(t, manifest.Engine) +} + +func TestInstallCancellationRemovesDownloadedArtifacts(t *testing.T) { + firstPayload := []byte("server") + secondStarted := make(chan struct{}, 1) + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/server.zip" { + _, _ = w.Write(firstPayload) + return + } + _, _ = w.Write([]byte("partial")) + if flusher, ok := w.(http.Flusher); ok { + flusher.Flush() + } + secondStarted <- struct{}{} + <-r.Context().Done() + })) + defer server.Close() + + marker := filepath.Join(t.TempDir(), "must-not-exist") + artifacts := []InstallArtifact{ + pinnedArtifact("server", server.URL+"/server.zip", firstPayload), + {Name: "cudart", URL: server.URL + "/cudart.zip", SHA256: strings.Repeat("c", 64)}, + } + manifest := artifactInstallManifest(t, "artifact-cancel", marker, artifacts) + executor := newTestExecutor(t, manifest) + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan error, 1) + go func() { + done <- executor.Install(ctx, manifest.Engine) + }() + + select { + case <-secondStarted: + cancel() + case <-time.After(5 * time.Second): + t.Fatal("second artifact download did not start") + } + select { + case err := <-done: + if !errors.Is(err, context.Canceled) { + t.Fatalf("install error = %v, want context cancellation", err) + } + case <-time.After(5 * time.Second): + t.Fatal("cancelled install did not return") + } + assertNoArtifactTemps(t, manifest.Engine) +} + +func pinnedArtifact(name, url string, payload []byte) InstallArtifact { + sum := sha256.Sum256(payload) + return InstallArtifact{Name: name, URL: url, SHA256: hex.EncodeToString(sum[:])} +} + +func artifactInstallManifest(t *testing.T, engine, marker string, artifacts []InstallArtifact) *Manifest { + t.Helper() + manifest := testEngineManifest(fakeEngineBin) + manifest.Engine = engine + manifest.DisplayName = "Artifact Test" + platform := manifest.Platforms[hostKey()] + platform.Detect = []string{marker} + platform.Install = &Install{ + Artifacts: artifacts, + Run: []string{fakeEngineBin, "captureargs", marker, "{download_server}", "{download_cudart}"}, + } + manifest.Platforms[hostKey()] = platform + if err := manifest.Validate(); err != nil { + t.Fatalf("validate fixture: %v", err) + } + return manifest +} + +func assertNoArtifactTemps(t *testing.T, engine string) { + t.Helper() + matches, err := filepath.Glob(filepath.Join(os.TempDir(), "nvpair-engine-"+engine+"-*")) + if err != nil { + t.Fatalf("glob temporary artifacts: %v", err) + } + if len(matches) != 0 { + t.Fatalf("temporary artifacts remain: %v", matches) + } +} diff --git a/services/nvpair-engine-manager/registry.go b/services/nvpair-engine-manager/registry.go index 720e3dcb..9f9b84fb 100644 --- a/services/nvpair-engine-manager/registry.go +++ b/services/nvpair-engine-manager/registry.go @@ -41,6 +41,8 @@ var allowedPlaceholders = map[string]bool{ var placeholderRe = regexp.MustCompile(`\{([a-zA-Z_][a-zA-Z0-9_]*)\}`) var resultMatchFieldPathRe = regexp.MustCompile(`^[a-zA-Z_][a-zA-Z0-9_-]*(?:\.[a-zA-Z_][a-zA-Z0-9_-]*)*$`) +var artifactNameRe = regexp.MustCompile(`^[a-z][a-z0-9_]{0,31}$`) +var sha256Re = regexp.MustCompile(`^[a-fA-F0-9]{64}$`) // engineNameRe restricts engine names to a safe charset — the name is // used as a filesystem path component (the per-engine install dir), so @@ -68,12 +70,13 @@ type Platform struct { Runtime Runtime `json:"runtime"` } -// Install describes how to obtain the engine in user mode. A pinned -// download (fetch.sha256 set) is checksum-verified before its `run` -// command executes; an unpinned fetch is HTTPS-only (see download). +// Install describes how to obtain the engine in user mode. Fetch may be +// unpinned, but every member of Artifacts is checksum-verified before `run` +// executes. All downloads are HTTPS-only outside loopback. type Install struct { - Fetch *Fetch `json:"fetch,omitempty"` - Run []string `json:"run,omitempty"` + Fetch *Fetch `json:"fetch,omitempty"` + Artifacts []InstallArtifact `json:"artifacts,omitempty"` + Run []string `json:"run,omitempty"` // Script is an escape hatch for vendors that only ship a script // installer. It runs without checksum verification — strictly opt-in // and logged as unpinned. Prefer fetch+run whenever the vendor publishes @@ -99,6 +102,14 @@ type Fetch struct { SHA256 string `json:"sha256"` } +// InstallArtifact is one checksum-pinned member of a multi-file install. +// Its download is available to install.run as {download_}. +type InstallArtifact struct { + Name string `json:"name"` + URL string `json:"url"` + SHA256 string `json:"sha256"` +} + // Runtime is how to launch + probe the engine. By default the launched // engine binds loopback; an engine may set Bind (substituted as {host}) // to listen elsewhere — inference engines use "0.0.0.0" to serve the @@ -643,15 +654,35 @@ func (p *Platform) validate(key string) error { return fmt.Errorf("platform %q: runtime.mode %q invalid (want \"process\" or \"command\")", key, p.Runtime.Mode) } if p.Install != nil { - if len(p.Install.Script) > 0 && (p.Install.Fetch != nil || len(p.Install.Run) > 0) { - return fmt.Errorf("platform %q: install.script is mutually exclusive with fetch/run (a script install cannot also be checksum-pinned)", key) + hasArtifacts := len(p.Install.Artifacts) > 0 + if len(p.Install.Script) > 0 && (p.Install.Fetch != nil || hasArtifacts || len(p.Install.Run) > 0) { + return fmt.Errorf("platform %q: install.script is mutually exclusive with fetch/artifacts/run (a script install cannot also be checksum-pinned)", key) + } + if p.Install.Fetch != nil && hasArtifacts { + return fmt.Errorf("platform %q: install.fetch and install.artifacts are mutually exclusive", key) } - if len(p.Install.Run) > 0 && p.Install.Fetch == nil { - return fmt.Errorf("platform %q: install.run requires a fetch (the artifact the run command unpacks)", key) + if len(p.Install.Run) > 0 && p.Install.Fetch == nil && !hasArtifacts { + return fmt.Errorf("platform %q: install.run requires a fetch or artifacts (the downloads the run command uses)", key) } if p.Install.Fetch != nil && strings.TrimSpace(p.Install.Fetch.URL) == "" { return fmt.Errorf("platform %q: install.fetch.url is required when fetch is present", key) } + names := make(map[string]bool, len(p.Install.Artifacts)) + for index, artifact := range p.Install.Artifacts { + if !artifactNameRe.MatchString(artifact.Name) { + return fmt.Errorf("platform %q: install.artifacts[%d].name %q must match [a-z][a-z0-9_]{0,31}", key, index, artifact.Name) + } + if names[artifact.Name] { + return fmt.Errorf("platform %q: duplicate install artifact name %q", key, artifact.Name) + } + names[artifact.Name] = true + if err := validateDownloadURL(artifact.URL); err != nil { + return fmt.Errorf("platform %q: install artifact %q: %w", key, artifact.Name, err) + } + if !sha256Re.MatchString(strings.TrimSpace(artifact.SHA256)) { + return fmt.Errorf("platform %q: install artifact %q requires a 64-character hexadecimal sha256", key, artifact.Name) + } + } switch p.Install.Mode { case "", "user", "admin": default: @@ -756,10 +787,22 @@ func (a *Action) validate(name string) error { // validatePlaceholders rejects any `{token}` outside allowedPlaceholders // across every templated string in the manifest. func (m *Manifest) validatePlaceholders() error { + allowed := make(map[string]bool, len(allowedPlaceholders)) + for name := range allowedPlaceholders { + allowed[name] = true + } + for _, platform := range m.Platforms { + if platform.Install == nil { + continue + } + for _, artifact := range platform.Install.Artifacts { + allowed["download_"+artifact.Name] = true + } + } for _, s := range m.templatedStrings() { for _, match := range placeholderRe.FindAllStringSubmatch(s, -1) { - if !allowedPlaceholders[match[1]] { - return fmt.Errorf("unknown placeholder {%s} (allowed: %s)", match[1], strings.Join(allowedPlaceholderList(), ", ")) + if !allowed[match[1]] { + return fmt.Errorf("unknown placeholder {%s} (allowed: %s)", match[1], strings.Join(placeholderList(allowed), ", ")) } } } @@ -769,8 +812,12 @@ func (m *Manifest) validatePlaceholders() error { // allowedPlaceholderList returns the allowed placeholder names, sorted, // so error messages can't drift from the actual allow-set. func allowedPlaceholderList() []string { - out := make([]string, 0, len(allowedPlaceholders)) - for k := range allowedPlaceholders { + return placeholderList(allowedPlaceholders) +} + +func placeholderList(placeholders map[string]bool) []string { + out := make([]string, 0, len(placeholders)) + for k := range placeholders { out = append(out, k) } sort.Strings(out) diff --git a/services/nvpair-engine-manager/registry_test.go b/services/nvpair-engine-manager/registry_test.go index 29102c99..8143b7e4 100644 --- a/services/nvpair-engine-manager/registry_test.go +++ b/services/nvpair-engine-manager/registry_test.go @@ -105,6 +105,29 @@ func TestValidateAcceptsUnpinnedFetch(t *testing.T) { } } +func TestValidateAcceptsNamedInstallArtifacts(t *testing.T) { + m := validManifest() + setInstallArtifacts(&m, validInstallArtifacts()) + if err := m.Validate(); err != nil { + t.Fatalf("named install artifacts rejected: %v", err) + } +} + +func validInstallArtifacts() []InstallArtifact { + return []InstallArtifact{ + {Name: "server", URL: "https://example/server.zip", SHA256: strings.Repeat("a", 64)}, + {Name: "cudart", URL: "https://example/cudart.zip", SHA256: strings.Repeat("b", 64)}, + } +} + +func setInstallArtifacts(m *Manifest, artifacts []InstallArtifact) { + p := m.Platforms["linux/amd64"] + p.Install.Fetch = nil + p.Install.Artifacts = artifacts + p.Install.Run = []string{"extract", "{download_server}", "{download_cudart}"} + m.Platforms["linux/amd64"] = p +} + func TestValidateRejectsBadEngineName(t *testing.T) { for _, bad := range []string{"../evil", "a/b", `a\b`, "..", ".", "a b", ""} { m := validManifest() @@ -149,6 +172,37 @@ func TestValidateRejects(t *testing.T) { p.Install.Fetch = nil m.Platforms["linux/amd64"] = p }, "requires a fetch"}, + {"fetch with artifacts", func(m *Manifest) { + p := m.Platforms["linux/amd64"] + p.Install.Artifacts = validInstallArtifacts() + m.Platforms["linux/amd64"] = p + }, "mutually exclusive"}, + {"invalid artifact name", func(m *Manifest) { + artifacts := validInstallArtifacts() + artifacts[0].Name = "../server" + setInstallArtifacts(m, artifacts) + }, "must match"}, + {"duplicate artifact name", func(m *Manifest) { + artifacts := validInstallArtifacts() + artifacts[1].Name = artifacts[0].Name + setInstallArtifacts(m, artifacts) + }, "duplicate install artifact"}, + {"insecure artifact URL", func(m *Manifest) { + artifacts := validInstallArtifacts() + artifacts[0].URL = "http://example.com/server.zip" + setInstallArtifacts(m, artifacts) + }, "must be https"}, + {"invalid artifact checksum", func(m *Manifest) { + artifacts := validInstallArtifacts() + artifacts[0].SHA256 = "deadbeef" + setInstallArtifacts(m, artifacts) + }, "64-character hexadecimal"}, + {"unknown artifact placeholder", func(m *Manifest) { + setInstallArtifacts(m, validInstallArtifacts()) + p := m.Platforms["linux/amd64"] + p.Install.Run = append(p.Install.Run, "{download_gpu}") + m.Platforms["linux/amd64"] = p + }, "unknown placeholder {download_gpu}"}, {"bad install mode", func(m *Manifest) { p := m.Platforms["linux/amd64"] p.Install.Mode = "root" From 092e1be7beda0b30f052d847c40040dccecbf50f Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 02:18:54 -0700 Subject: [PATCH 09/70] feat(llamacpp): register the opt-in backend Signed-off-by: Sherief Farouk --- .../launch_controls_test.go | 13 +- services/nvpair-engine-manager/launch_test.go | 14 ++- .../manifests/llamacpp.json | 117 ++++++++++++++++++ .../nvpair-engine-manager/remediation_test.go | 2 +- services/nvpair-proxy/engines.go | 15 +++ services/nvpair-proxy/engines_test.go | 37 +++++- services/nvpair-ui-broker/engineproxy.go | 4 + services/nvpair-ui-broker/engineproxy_test.go | 4 +- services/nvpair-ui-broker/enginesettings.go | 68 +++++----- .../enginesettings_recovery.go | 13 +- .../enginesettings_review_test.go | 22 ++-- .../nvpair-ui-broker/enginesettings_test.go | 16 +-- .../health_connections_test.go | 1 + services/shared/engines/engines.go | 13 +- services/shared/engines/engines_test.go | 2 +- services/shared/noderec/noderec.go | 5 +- services/shared/noderec/noderec_test.go | 1 + 17 files changed, 270 insertions(+), 77 deletions(-) create mode 100644 services/nvpair-engine-manager/manifests/llamacpp.json diff --git a/services/nvpair-engine-manager/launch_controls_test.go b/services/nvpair-engine-manager/launch_controls_test.go index f8718996..8c9f12ed 100644 --- a/services/nvpair-engine-manager/launch_controls_test.go +++ b/services/nvpair-engine-manager/launch_controls_test.go @@ -109,7 +109,7 @@ func TestSavedControlsCannotBypassLaunchValidation(t *testing.T) { func TestBundledNetworkingControls(t *testing.T) { reg := loadWithOverrides(t, t.TempDir()) // Adding a bundled engine requires an explicit networking review and cases. - wantEngines := []string{"lmstudio", "ollama"} + wantEngines := []string{"llamacpp", "lmstudio", "ollama"} names := reg.Names() slices.Sort(names) if !slices.Equal(names, wantEngines) { @@ -126,13 +126,20 @@ func TestBundledNetworkingControls(t *testing.T) { t.Fatal("missing reviewed networking controls") } var valid, invalid []string - if name == "lmstudio" { + switch name { + case "lmstudio": if !reflect.DeepEqual(policy.Controls, []LaunchControl{{Value: "{server.port}", Flags: []string{"--port", "-p"}}, {Value: "{server.host}", Flags: []string{"--bind"}, Env: []string{"LMS_SERVER_HOST"}}, {Value: "{cors.enabled}", Implicit: implicitLaunchValue("true"), Flags: []string{"--cors"}}}) { t.Fatal("incomplete LM Studio controls") } valid = []string{"--port 23456", "--port=23456", "-p 23456", "-p23456", "-p=23456", `"-p" "23456"`, "--port 23456 -p23456", "-- -p23456"} invalid = []string{"-p0", "-p65536", "-p", "-pno", "--port 23456 -p23457", "-vp23456", "-vp=23456", "--bind 0.0.0.0", "--bind=::", "LMS_SERVER_HOST=0.0.0.0", "--cors=false", "--cors=true", "--cors=", "-- --bind 0.0.0.0"} - } else { + case "llamacpp": + if !reflect.DeepEqual(policy.Controls, []LaunchControl{{Value: "{server.host}", Flags: []string{"--host"}}, {Value: "{server.port}", Flags: []string{"--port"}}, {Value: "{cors.origins}", Flags: []string{"--cors-origins"}}}) { + t.Fatal("incomplete llama.cpp controls") + } + valid = []string{"--host 127.0.0.1 --port 23456", "--host=127.0.0.1 --port=23456", "--port 23456 --cors-origins https://example.test"} + invalid = []string{"--host 0.0.0.0", "--host=::", "--port 0", "--port 65536", "--port", "--cors-origins=*", "--port 23456 --port 23457", "-- --host 0.0.0.0"} + default: if !reflect.DeepEqual(policy.Controls, []LaunchControl{{Value: "{server.host}:{server.port}", Env: []string{"OLLAMA_HOST"}}, {Value: "{cors.origins}", Env: []string{"OLLAMA_ORIGINS"}}}) { t.Fatal("incomplete Ollama controls") } diff --git a/services/nvpair-engine-manager/launch_test.go b/services/nvpair-engine-manager/launch_test.go index d2b8eb32..c3389868 100644 --- a/services/nvpair-engine-manager/launch_test.go +++ b/services/nvpair-engine-manager/launch_test.go @@ -22,7 +22,7 @@ func (command launchCommand) text() (string, error) { func TestResolvedLaunchMatchesBundledEngines(t *testing.T) { reg := loadWithOverrides(t, t.TempDir()) vars := map[string]string{"host": "127.0.0.1", "port": "12345", "cli": "/test path/lms", "install_dir": "/test path"} - for _, engine := range []string{"ollama", "lmstudio"} { + for _, engine := range []string{"ollama", "lmstudio", "llamacpp"} { manifest, ok := reg.Get(engine) if !ok { t.Fatalf("missing bundled engine %q", engine) @@ -40,9 +40,19 @@ func TestResolvedLaunchMatchesBundledEngines(t *testing.T) { if strings.HasPrefix(platformKey, "linux/") { want = append([]string{"LD_LIBRARY_PATH=/test path/lib/ollama"}, want...) } - } else { + } else if engine == "lmstudio" { launch, err = resolveCommandLaunch(platform.Runtime.Start[0], vars) want = []string{"/test path/lms", "server", "start", "--port", "12345", "--bind", "127.0.0.1"} + } else { + var bin string + bin, err = resolvePlaceholders(platform.Runtime.Bin, vars) + if err == nil { + launch, err = resolveProcessLaunch(platform.Runtime, bin, vars) + } + want = []string{"LLAMA_CACHE=/test path-models", bin, "--host", "127.0.0.1", "--port", "12345"} + if strings.HasPrefix(platformKey, "linux/") { + want = append([]string{"LD_LIBRARY_PATH=/test path/build/bin"}, want...) + } } if err != nil { t.Fatal(err) diff --git a/services/nvpair-engine-manager/manifests/llamacpp.json b/services/nvpair-engine-manager/manifests/llamacpp.json new file mode 100644 index 00000000..04d4cc5f --- /dev/null +++ b/services/nvpair-engine-manager/manifests/llamacpp.json @@ -0,0 +1,117 @@ +{ + "$schema": "../manifest.schema.json", + "engine": "llamacpp", + "display_name": "llama.cpp", + "manifest_version": 1, + "install": { "mode": "user" }, + "runtime": { + "editable_launch": { + "fixed_args": [], + "controls": [ + { "flags": ["--host"], "value": "{server.host}" }, + { "flags": ["--port"], "value": "{server.port}" }, + { "flags": ["--cors-origins"], "value": "{cors.origins}" } + ] + }, + "args": ["--host", "{host}", "--port", "{port}"], + "env": { "LLAMA_CACHE": "{install_dir}-models" }, + "bind": "127.0.0.1", + "port": 8081, + "ready": { "http": "http://127.0.0.1:{port}/health", "status": 200, "timeout_s": 60 }, + "stop": { "signal": "term", "grace_s": 10 }, + "health": { "http": "http://127.0.0.1:{port}/health", "status": 200, "interval_s": 5 } + }, + "platforms": { + "windows/amd64": { + "detect": ["{install_dir}\\llama-server.exe"], + "install": { + "artifacts": [ + { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-win-cuda-12.4-x64.zip", "sha256": "3c806a6ceccc3dae1c743ceb1a1fb2cce5b76f40bfbd4c6b7b8afb6ef45a5807" }, + { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-bin-win-cuda-12.4-x64.zip", "sha256": "8c79a9b226de4b3cacfd1f83d24f962d0773be79f1e7b75c6af4ded7e32ae1d6" } + ], + "run": ["powershell.exe", "-NoProfile", "-NonInteractive", "-Command", "& { param($server, $cudart, $destination) $ErrorActionPreference = 'Stop'; Expand-Archive -LiteralPath $server -DestinationPath $destination -Force; Expand-Archive -LiteralPath $cudart -DestinationPath $destination -Force }", "{download_server}", "{download_cudart}", "{install_dir}"] + }, + "uninstall": { "run": ["cmd", "/c", "rmdir", "/s", "/q", "{install_dir}"] }, + "runtime": { "bin": "{install_dir}\\llama-server.exe" } + }, + "windows/arm64": { + "detect": ["{install_dir}\\llama-server.exe"], + "install": { + "artifacts": [ + { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-win-cuda-13.4-arm64.zip", "sha256": "a4060b5031a0e862e225d4a7c4aa403852ebedf598906ef8258a86cb77de8351" }, + { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-bin-win-cuda-13.4-arm64.zip", "sha256": "642dcde8805b3e3165ca710a5443b3b4044b27d96bd3ee3132473988c9bcb774" } + ], + "run": ["powershell.exe", "-NoProfile", "-NonInteractive", "-Command", "& { param($server, $cudart, $destination) $ErrorActionPreference = 'Stop'; Expand-Archive -LiteralPath $server -DestinationPath $destination -Force; Expand-Archive -LiteralPath $cudart -DestinationPath $destination -Force }", "{download_server}", "{download_cudart}", "{install_dir}"] + }, + "uninstall": { "run": ["cmd", "/c", "rmdir", "/s", "/q", "{install_dir}"] }, + "runtime": { "bin": "{install_dir}\\llama-server.exe" } + }, + "linux/amd64": { + "detect": ["{install_dir}/build/bin/llama-server"], + "install": { + "artifacts": [ + { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-ubuntu-cuda-12.8-x64.tar.gz", "sha256": "c2ab9e19838513ff69d1af8d999ad717dd3c7ee4714ac04c7ed5ab9077c50e4e" }, + { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-b11146-bin-ubuntu-cuda-12.8-x64.tar.gz", "sha256": "1466daea60aad1144819e151b2bae19d54556cf1da6c129c4f55a5ded2637c25" } + ], + "run": ["sh", "-c", "tar -xzf \"$1\" -C \"$3\" && tar -xzf \"$2\" -C \"$3\"", "llamacpp-install", "{download_server}", "{download_cudart}", "{install_dir}"] + }, + "uninstall": { "run": ["rm", "-rf", "{install_dir}"] }, + "runtime": { "bin": "{install_dir}/build/bin/llama-server", "env": { "LD_LIBRARY_PATH": "{install_dir}/build/bin" } } + }, + "linux/arm64": { + "detect": ["{install_dir}/build/bin/llama-server"], + "install": { + "artifacts": [ + { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-ubuntu-cuda-13.4-arm64.tar.gz", "sha256": "4e00496ab6cdee9c00afb11de3cb9d10f9da7e17147d8ed14ca3af05209b400f" }, + { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-b11146-bin-ubuntu-cuda-13.4-arm64.tar.gz", "sha256": "7f46057efcba6338c58ed9f91c06f268c290f3a228fdbdd4bea229dd60b0e094" } + ], + "run": ["sh", "-c", "tar -xzf \"$1\" -C \"$3\" && tar -xzf \"$2\" -C \"$3\"", "llamacpp-install", "{download_server}", "{download_cudart}", "{install_dir}"] + }, + "uninstall": { "run": ["rm", "-rf", "{install_dir}"] }, + "runtime": { "bin": "{install_dir}/build/bin/llama-server", "env": { "LD_LIBRARY_PATH": "{install_dir}/build/bin" } } + }, + "darwin/arm64": { + "detect": ["{install_dir}/build/bin/llama-server"], + "install": { + "fetch": { "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-macos-arm64.tar.gz", "sha256": "1ad3f9eff80edb9dbef4259ad564d1720612ef7eea48fa4afed0e54f5f3d5711" }, + "run": ["tar", "-xzf", "{download}", "-C", "{install_dir}"] + }, + "uninstall": { "run": ["rm", "-rf", "{install_dir}"] }, + "runtime": { "bin": "{install_dir}/build/bin/llama-server" } + }, + "darwin/amd64": { + "detect": ["{install_dir}/build/bin/llama-server"], + "install": { + "fetch": { "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-macos-x64.tar.gz", "sha256": "305f0e3a17d2c01eb205cd0a62128357f1ec3b55329cb084d94e5ec0115d7a3b" }, + "run": ["tar", "-xzf", "{download}", "-C", "{install_dir}"] + }, + "uninstall": { "run": ["rm", "-rf", "{install_dir}"] }, + "runtime": { "bin": "{install_dir}/build/bin/llama-server" } + } + }, + "actions": { + "list_models": { + "description": "List models available in the managed llama.cpp cache.", + "http": { "method": "GET", "path": "/models" }, + "result": { "array": "data", "field": "id" } + }, + "loaded_models": { + "description": "List models currently resident in memory.", + "http": { "method": "GET", "path": "/models" }, + "result": { "array": "data", "field": "id", "match": { "field": "status.value", "in": ["loaded"] } } + }, + "pull_model": { + "description": "Download a model into the managed cache (params: {\"model\": \"\"}).", + "http": { "method": "POST", "path": "/models", "body_schema": { "model": "string" } }, + "progress_protocol": "llamacpp-models-sse" + }, + "load_model": { + "description": "Load a cached model into memory (params: {\"model\": \"\"}).", + "http": { "method": "POST", "path": "/models/load", "body_schema": { "model": "string" } } + }, + "unload_model": { + "description": "Unload a model from memory (params: {\"model\": \"\"}).", + "http": { "method": "POST", "path": "/models/unload", "body_schema": { "model": "string" } } + } + } +} diff --git a/services/nvpair-engine-manager/remediation_test.go b/services/nvpair-engine-manager/remediation_test.go index 1999373b..3d320698 100644 --- a/services/nvpair-engine-manager/remediation_test.go +++ b/services/nvpair-engine-manager/remediation_test.go @@ -453,7 +453,7 @@ func TestBundledManifestsGolden(t *testing.T) { if err := reg.LoadFS(bundledManifests, "manifests"); err != nil { t.Fatalf("bundled manifests invalid: %v", err) } - for _, want := range []string{"ollama", "lmstudio"} { + for _, want := range []string{"ollama", "lmstudio", "llamacpp"} { m, ok := reg.Get(want) if !ok { t.Fatalf("missing bundled engine %q (have %v)", want, reg.Names()) diff --git a/services/nvpair-proxy/engines.go b/services/nvpair-proxy/engines.go index 2a5054b2..2c1d0900 100644 --- a/services/nvpair-proxy/engines.go +++ b/services/nvpair-proxy/engines.go @@ -135,6 +135,12 @@ var lmStudioBaseRoutes = []route{ {Path: "/v1/models", Role: roleModelListOpenAIGET}, } +// llamaCPPBaseRoutes maps the facade's OpenAI-compatible model-list path to +// llama.cpp's router endpoint. The response already uses the OpenAI envelope. +var llamaCPPBaseRoutes = []route{ + {Path: "/v1/models", UpstreamPath: "/models", Role: roleModelListOpenAIGET}, +} + // openAIInferenceRoutes is the OpenAI-compatible inference surface. var openAIInferenceRoutes = []route{ {Path: "/v1/chat/completions", Role: roleInferencePOST}, @@ -152,8 +158,10 @@ var profiles = buildProfiles() func buildProfiles() []engineProfile { ollama, _ := engines.ByName("ollama") lmstudio, _ := engines.ByName("lmstudio") + llamacpp, _ := engines.ByName("llamacpp") ollamaRoutes := slices.Concat(ollamaBaseRoutes, openAIInferenceRoutes, anthropicInferenceRoutes) lmStudioRoutes := slices.Concat(lmStudioBaseRoutes, openAIInferenceRoutes, anthropicInferenceRoutes) + llamaCPPRoutes := slices.Concat(llamaCPPBaseRoutes, openAIInferenceRoutes) return []engineProfile{ { @@ -174,6 +182,13 @@ func buildProfiles() []engineProfile { // stored value predates the current default of 1234. ReservedPersistedPort: 1235, }, + { + Engine: llamacpp, + StandalonePort: 8080, + Routes: llamaCPPRoutes, + ModelNaming: exactID, + ReservedPersistedPort: 8081, + }, } } diff --git a/services/nvpair-proxy/engines_test.go b/services/nvpair-proxy/engines_test.go index 422159c7..feaf79b3 100644 --- a/services/nvpair-proxy/engines_test.go +++ b/services/nvpair-proxy/engines_test.go @@ -3,7 +3,42 @@ package main -import "testing" +import ( + "slices" + "testing" + + "nvpair-shared/engines" +) + +func TestProfilesMatchSharedEngines(t *testing.T) { + names := make([]string, len(profiles)) + for i, profile := range profiles { + names[i] = profile.Name + } + if !slices.Equal(names, engines.Names()) { + t.Fatalf("proxy profiles = %v, want canonical engines %v", names, engines.Names()) + } +} + +func TestLlamaCPPProfile(t *testing.T) { + profile, ok := profileFor("llamacpp") + if !ok { + t.Fatal("llamacpp profile missing") + } + route, ok := profile.routeFor("GET", "/v1/models") + if !ok || route.Role != roleModelListOpenAIGET || route.upstreamPath() != "/models" { + t.Fatalf("model list route = %+v, %v", route, ok) + } + if role, ok := profile.roleFor("POST", "/v1/chat/completions"); !ok || role != roleInferencePOST { + t.Fatalf("chat route = %v, %v", role, ok) + } + if got := profile.normalizeModel("org/model:Q4_K_M"); got != "org/model:Q4_K_M" { + t.Fatalf("exact model id normalized to %q", got) + } + if profile.StandalonePort != 8080 || profile.ReservedPersistedPort != 8081 { + t.Fatalf("ports = facade %d, reserved %d", profile.StandalonePort, profile.ReservedPersistedPort) + } +} // Routes is a classifier, not an allowlist. handlePlain forwards every // loopback path into handleHTTP with no filtering, so a path the table does diff --git a/services/nvpair-ui-broker/engineproxy.go b/services/nvpair-ui-broker/engineproxy.go index d37823de..0163bd67 100644 --- a/services/nvpair-ui-broker/engineproxy.go +++ b/services/nvpair-ui-broker/engineproxy.go @@ -171,6 +171,7 @@ func buildEngineProxyProfiles() []engineProxyProfile { // LM Studio is the one engine engine-manager may move while running: // its identified command-mode runtime has an official stop command. "lmstudio": {Ownership: managedEngine, HealthProbePath: "/v1/models"}, + "llamacpp": {Ownership: prepositionedEngine, HealthProbePath: "/health"}, } out := make([]engineProxyProfile, 0, len(engines.All())) for _, e := range engines.All() { @@ -423,6 +424,9 @@ func (b *Broker) prepareEnabledFacades() { } func (b *Broker) preparePrepositionedFacade(profile engineProxyProfile) { + if b.prepareExplicitEngineSettings(profile.Name) { + return + } b.engineProxy(profile).backendPort.Store(int32(profile.EnginePortBase)) } diff --git a/services/nvpair-ui-broker/engineproxy_test.go b/services/nvpair-ui-broker/engineproxy_test.go index 1c650063..09589c29 100644 --- a/services/nvpair-ui-broker/engineproxy_test.go +++ b/services/nvpair-ui-broker/engineproxy_test.go @@ -46,6 +46,7 @@ func TestEngineHealthProbePaths(t *testing.T) { }{ {"ollama", "/"}, {"lmstudio", "/v1/models"}, + {"llamacpp", "/health"}, } { p, ok := engineProxyProfileFor(tc.engine) if !ok { @@ -92,7 +93,7 @@ func TestBrokerConstantsMatchTheEngineTable(t *testing.T) { } } -// Ownership is the one judgment call in adding an engine, so the two values in +// Ownership is the one judgment call in adding an engine, so every value in // the table today are pinned explicitly. Getting these backwards does not fail // to compile — it silently changes which engine the broker believes it may stop. func TestEngineOwnershipAssignments(t *testing.T) { @@ -102,6 +103,7 @@ func TestEngineOwnershipAssignments(t *testing.T) { }{ {"ollama", adoptedEngine}, {"lmstudio", managedEngine}, + {"llamacpp", prepositionedEngine}, } { p, ok := engineProxyProfileFor(tc.engine) if !ok { diff --git a/services/nvpair-ui-broker/enginesettings.go b/services/nvpair-ui-broker/enginesettings.go index 9f65e88b..de6d2bbe 100644 --- a/services/nvpair-ui-broker/enginesettings.go +++ b/services/nvpair-ui-broker/enginesettings.go @@ -18,6 +18,7 @@ import ( "nvpair-shared/appdir" "nvpair-shared/clustertrust" + "nvpair-shared/engines" settings "nvpair-shared/enginesettings" "nvpair-shared/noderec" ) @@ -160,17 +161,24 @@ func (b *Broker) settingsWorkerCall(ctx context.Context, method string, params a } func (b *Broker) settingsProxy(engine string) *proxyProcess { - if engine == "ollama" { - return b.getProxy() + profile, ok := engineProxyProfileFor(engine) + if !ok { + return nil } - if engine == "lmstudio" { - return b.getLMStudioProxy() + return b.engineProxyHandle(profile) +} + +func isEngineDiscoveryService(service noderec.ServiceKey) bool { + for _, profile := range engineProxyProfiles { + if profile.DiscoveryService == service { + return true + } } - return nil + return false } func (b *Broker) settingsSnapshotLocked(ctx context.Context, engine string) (settings.Snapshot, error) { - if engine != "ollama" && engine != "lmstudio" { + if _, ok := engineProxyProfileFor(engine); !ok { return settings.Snapshot{}, fmt.Errorf("this engine does not support settings") } if err := b.loadEngineSettingsLocked(); err != nil { @@ -233,8 +241,8 @@ func (b *Broker) settingsSnapshotLocked(ctx context.Context, engine string) (set func (b *Broker) publishSettingsLocked() { all := make([]settings.Snapshot, 0, len(b.engineSettings)) - for _, engine := range []string{"ollama", "lmstudio"} { - if record := b.engineSettings[engine]; record != nil { + for _, profile := range engineProxyProfiles { + if record := b.engineSettings[profile.Name]; record != nil { record.Snapshot.Sequence++ all = append(all, record.Snapshot) if b.codec != nil { @@ -261,7 +269,7 @@ func (b *Broker) validateSettingsPortsLocked(ctx context.Context, engine string, // Include every registered PAIR listener, including services added later. if b.regCache != nil { for _, service := range b.regCache.Snapshot() { - if service.Port > 0 && service.Service != noderec.ServiceOllama && service.Service != noderec.ServiceLMStudio { + if service.Port > 0 && !isEngineDiscoveryService(service.Service) { reserved[service.Port] = true } } @@ -283,18 +291,18 @@ func (b *Broker) validateSettingsPortsLocked(ctx context.Context, engine string, return fmt.Errorf("a selected port is reserved by another configured engine") } } - for _, other := range []string{"ollama", "lmstudio"} { - if other == engine { + for _, profile := range engineProxyProfiles { + if profile.Name == engine { continue } - if record := b.engineSettings[other]; record != nil { + if record := b.engineSettings[profile.Name]; record != nil { port := record.Snapshot.Settings.ProxyPort if port == config.ServerPort || port == config.ProxyPort { return fmt.Errorf("a selected port is reserved by another configured proxy") } } - if proxy := b.settingsProxy(other); proxy != nil { - _, port := proxy.Status(other) + if proxy := b.settingsProxy(profile.Name); proxy != nil { + _, port := proxy.Status(profile.Name) if port == config.ServerPort || port == config.ProxyPort { return fmt.Errorf("a selected port is already used by another proxy") } @@ -350,10 +358,8 @@ func (b *Broker) runSettingsOperationLocked(ctx context.Context, engine string, if err := b.migrateSettingsArgumentsLocked(ctx, engine, record); err != nil { return err } - service := noderec.ServiceOllama - if engine == "lmstudio" { - service = noderec.ServiceLMStudio - } + profile, _ := engineProxyProfileFor(engine) + service := profile.DiscoveryService if b.regCache != nil { b.unregisterService(service) } @@ -398,11 +404,7 @@ func (b *Broker) runSettingsOperationLocked(ctx context.Context, engine string, record.Snapshot.EffectiveServerPort = launch.EffectivePort record.Snapshot.Editable = launch.Editable record.Snapshot.Reason = launch.Reason - if engine == "ollama" { - b.ollamaState().backendPort.Store(int32(launch.EffectivePort)) - } else { - b.lmstudioState().backendPort.Store(int32(launch.EffectivePort)) - } + b.engineProxy(profile).backendPort.Store(int32(launch.EffectivePort)) } proxyReady := false if proxy := b.settingsProxy(engine); proxy != nil { @@ -683,12 +685,9 @@ func (b *Broker) rebindSettingsProxy(engine string, port int) error { // Disable automatic facade takeover before the ready event can race the // explicit rebind. The accepted journal restores these choices after restart. profile, _ := engineProxyProfileFor(engine) - b.engineProxy(profile).explicitSettings.Store(true) - if engine == "ollama" { - b.ollamaState().managedFacade.Store(false) - } else { - b.lmstudioState().managedFacade.Store(false) - } + state := b.engineProxy(profile) + state.explicitSettings.Store(true) + state.managedFacade.Store(false) _, rpcErr, err := p.Call(context.Background(), engine+":set-port", settingsJSON(map[string]int{"port": port})) if err != nil { return err @@ -696,13 +695,8 @@ func (b *Broker) rebindSettingsProxy(engine string, port int) error { if rpcErr != nil { return fmt.Errorf("%s", rpcErr.Message) } - if engine == "ollama" { - b.ollamaState().managedFacade.Store(false) - b.ollamaState().startupPort.Store(int32(port)) - } else { - b.lmstudioState().managedFacade.Store(false) - b.lmstudioState().startupPort.Store(int32(port)) - } + state.managedFacade.Store(false) + state.startupPort.Store(int32(port)) return nil } @@ -723,7 +717,7 @@ func (b *Broker) refreshEngineSettings(ctx context.Context) { readCtx, cancel := context.WithTimeout(ctx, 5*time.Second) defer cancel() changed := false - for _, engine := range []string{"ollama", "lmstudio"} { + for _, engine := range engines.Names() { var before settings.Snapshot if r := b.engineSettings[engine]; r != nil { before = r.Snapshot diff --git a/services/nvpair-ui-broker/enginesettings_recovery.go b/services/nvpair-ui-broker/enginesettings_recovery.go index ec444c39..dcaba5b1 100644 --- a/services/nvpair-ui-broker/enginesettings_recovery.go +++ b/services/nvpair-ui-broker/enginesettings_recovery.go @@ -73,17 +73,14 @@ func (b *Broker) prepareExplicitEngineSettings(engine string) bool { return false } profile, _ := engineProxyProfileFor(engine) - b.engineProxy(profile).explicitSettings.Store(true) + state := b.engineProxy(profile) + state.explicitSettings.Store(true) + state.managedFacade.Store(false) + state.backendPort.Store(int32(config.ServerPort)) + state.startupPort.Store(int32(config.ProxyPort)) if engine == "ollama" { - b.ollamaState().managedFacade.Store(false) b.managedOllamaBackend.Store(0) - b.ollamaState().backendPort.Store(int32(config.ServerPort)) - b.ollamaState().startupPort.Store(int32(config.ProxyPort)) b.syncCurrentEngineOllamaHostAliasReservation() - } else { - b.lmstudioState().managedFacade.Store(false) - b.lmstudioState().backendPort.Store(int32(config.ServerPort)) - b.lmstudioState().startupPort.Store(int32(config.ProxyPort)) } return true } diff --git a/services/nvpair-ui-broker/enginesettings_review_test.go b/services/nvpair-ui-broker/enginesettings_review_test.go index 50cb11be..33418945 100644 --- a/services/nvpair-ui-broker/enginesettings_review_test.go +++ b/services/nvpair-ui-broker/enginesettings_review_test.go @@ -21,8 +21,10 @@ func TestSettingsRebindAddressesOnlyRequestedFacade(t *testing.T) { t.Run(profile.Name, func(t *testing.T) { h := newSettingsHarness(t) p := h.b.getProxy() - _, ollamaBefore := p.Status("ollama") - _, lmstudioBefore := p.Status("lmstudio") + before := make(map[string]int, len(engineProxyProfiles)) + for _, candidate := range engineProxyProfiles { + _, before[candidate.Name] = p.Status(candidate.Name) + } ln, err := net.Listen("tcp", "127.0.0.1:0") if err != nil { t.Fatal(err) @@ -32,13 +34,8 @@ func TestSettingsRebindAddressesOnlyRequestedFacade(t *testing.T) { if err := h.b.rebindSettingsProxy(profile.Name, port); err != nil { t.Fatal(err) } - wantOllama, wantLMStudio := ollamaBefore, lmstudioBefore - if profile.Name == "ollama" { - wantOllama = port - } else { - wantLMStudio = port - } - for engine, want := range map[string]int{"ollama": wantOllama, "lmstudio": wantLMStudio} { + before[profile.Name] = port + for engine, want := range before { ready, got := p.Status(engine) if !ready || got != want { t.Fatalf("%s ready=%v port=%d, want %d", engine, ready, got, want) @@ -61,10 +58,13 @@ func TestExplicitSettingsBindFailurePreservesChosenPort(t *testing.T) { t.Fatal("explicit settings were not restored") } failure := settingsJSON(map[string]any{"code": "bind-failed", "port": requested}) - if profile.Name == "ollama" { + switch profile.Name { + case "ollama": b.forwardProxyNotification("error", failure) - } else { + case "lmstudio": b.forwardLMStudioProxyNotification("error", failure) + default: + b.forwardPrepositionedProxyNotification(profile, profile.addressed("error"), failure) } if got := b.engineProxy(profile).startupPort.Load(); got != requested { t.Fatalf("bind notification changed chosen port to %d", got) diff --git a/services/nvpair-ui-broker/enginesettings_test.go b/services/nvpair-ui-broker/enginesettings_test.go index e8d487b6..1344787b 100644 --- a/services/nvpair-ui-broker/enginesettings_test.go +++ b/services/nvpair-ui-broker/enginesettings_test.go @@ -16,6 +16,7 @@ import ( "testing" "time" + "nvpair-shared/engines" settings "nvpair-shared/enginesettings" "nvpair-shared/noderec" "nvpair-ui-broker/relay" @@ -38,7 +39,7 @@ type settingsHarness struct { func newSettingsHarness(t *testing.T) *settingsHarness { t.Helper() - ports := make([]int, 3) + ports := make([]int, 4) listeners := []net.Listener{} for i := range ports { ln, err := net.Listen("tcp", "127.0.0.1:0") @@ -58,9 +59,11 @@ func newSettingsHarness(t *testing.T) *settingsHarness { proxy := &proxyProcess{peer: proxyWorker.peer, facadeState: map[string]proxyFacadeState{ "ollama": {ready: true, port: ports[1]}, "lmstudio": {ready: true, port: ports[2]}, + "llamacpp": {ready: true, port: ports[3]}, }} - h.b.setProxy(proxy) - h.b.setLMStudioProxy(proxy) + for _, profile := range engineProxyProfiles { + h.b.setEngineProxyHandle(profile, proxy) + } go func() { for { msg, err := proxyCodec.Read() @@ -70,7 +73,8 @@ func newSettingsHarness(t *testing.T) *settingsHarness { if !msg.IsRequest() { continue } - if msg.Method != "ollama:set-port" && msg.Method != "lmstudio:set-port" { + engine, method := engines.SplitAddressedMethod(msg.Method) + if engine == "" || method != "set-port" { _ = proxyCodec.Respond(msg.ID, map[string]bool{"ok": true}) continue } @@ -79,10 +83,6 @@ func newSettingsHarness(t *testing.T) *settingsHarness { } _ = json.Unmarshal(msg.Params, &p) proxy.readyMu.Lock() - engine := "ollama" - if msg.Method == "lmstudio:set-port" { - engine = "lmstudio" - } proxy.facadeState[engine] = proxyFacadeState{ready: true, port: p.Port} proxy.readyMu.Unlock() _ = proxyCodec.Respond(msg.ID, map[string]int{"port": p.Port}) diff --git a/services/nvpair-ui-broker/health_connections_test.go b/services/nvpair-ui-broker/health_connections_test.go index 74e7231a..da685cba 100644 --- a/services/nvpair-ui-broker/health_connections_test.go +++ b/services/nvpair-ui-broker/health_connections_test.go @@ -51,4 +51,5 @@ func TestHealthChecksReuseConnections(t *testing.T) { test("ollama", "/", ollamaProxyProfile) test("lmstudio", "/v1/models", lmstudioProxyProfile) + test("llamacpp", "/health", mustEngineProxyProfile("llamacpp")) } diff --git a/services/shared/engines/engines.go b/services/shared/engines/engines.go index f3180677..ae2a9872 100644 --- a/services/shared/engines/engines.go +++ b/services/shared/engines/engines.go @@ -92,8 +92,8 @@ type Engine struct { // claims in managed mode. FacadePort int - // EnginePortBase is where PAIR relocates the engine so the proxy can take - // FacadePort, and the base of the next-free-port search. + // EnginePortBase is where PAIR runs or relocates the engine so the proxy can + // take FacadePort, and the base of any next-free-port search. EnginePortBase int // PortFile is the per-user file this engine's proxy persists its chosen @@ -149,6 +149,15 @@ var all = []Engine{ PortFile: "lmstudio-proxy-port.json", ProxyEnabledByDefault: true, }, + { + Name: "llamacpp", + DisplayName: "llama.cpp", + DiscoveryService: noderec.ServiceLlamaCPP, + FacadePort: 8080, + EnginePortBase: 8081, + PortFile: "llamacpp-proxy-port.json", + ProxyEnabledByDefault: false, + }, } // All returns the engine set in preparation order. The result is a copy, so a diff --git a/services/shared/engines/engines_test.go b/services/shared/engines/engines_test.go index 263e1393..5020586e 100644 --- a/services/shared/engines/engines_test.go +++ b/services/shared/engines/engines_test.go @@ -23,7 +23,7 @@ func TestOllamaIsPreparedFirst(t *testing.T) { func TestNames(t *testing.T) { got := Names() - want := []string{"ollama", "lmstudio"} + want := []string{"ollama", "lmstudio", "llamacpp"} if len(got) != len(want) { t.Fatalf("Names() = %v, want %v", got, want) } diff --git a/services/shared/noderec/noderec.go b/services/shared/noderec/noderec.go index 112e7fe3..f44a263d 100644 --- a/services/shared/noderec/noderec.go +++ b/services/shared/noderec/noderec.go @@ -10,7 +10,7 @@ // whose TXT map carries a schema version, the node's identity, its LAN address, // and one compact key per local service port, e.g.: // -// v=1;uuid=;cluster-uuid=;ip=192.168.1.10;ni=14318;ol=11434;lm=1234;er=14319;wl=14320;cl=14321;em=14322 +// v=1;uuid=;cluster-uuid=;ip=192.168.1.10;ni=14318;ol=11434;lm=1234;lc=8080;er=14319;wl=14320;cl=14321;em=14322 // // Design decisions this package encodes: // - SRV port is a fixed, NON-authoritative constant; consumers ignore it and @@ -87,6 +87,7 @@ const ( ServiceNodeInfo ServiceKey = "ni" ServiceOllama ServiceKey = "ol" ServiceLMStudio ServiceKey = "lm" + ServiceLlamaCPP ServiceKey = "lc" ServiceErrors ServiceKey = "er" ServiceWorkload ServiceKey = "wl" ServiceCluster ServiceKey = "cl" @@ -104,7 +105,7 @@ const ( // serviceKeyOrder is the deterministic emit order for service ports in TXT. var serviceKeyOrder = []ServiceKey{ - ServiceNodeInfo, ServiceOllama, ServiceLMStudio, + ServiceNodeInfo, ServiceOllama, ServiceLMStudio, ServiceLlamaCPP, ServiceErrors, ServiceWorkload, ServiceCluster, ServiceEngineManager, ServiceEngineControl, } diff --git a/services/shared/noderec/noderec_test.go b/services/shared/noderec/noderec_test.go index 8b0757e5..6a7bc1bf 100644 --- a/services/shared/noderec/noderec_test.go +++ b/services/shared/noderec/noderec_test.go @@ -147,6 +147,7 @@ func TestTransportPolicy(t *testing.T) { {ServiceNodeInfo, TransportPlain, false, false}, {ServiceOllama, TransportPlain, false, false}, {ServiceLMStudio, TransportPlain, false, false}, + {ServiceLlamaCPP, TransportPlain, false, false}, {ServiceEngineManager, TransportPlain, false, false}, {ServiceErrors, TransportMTLSWhenClustered, true, false}, {ServiceWorkload, TransportMTLSWhenClustered, true, false}, From ea6890fdbb735208de92f8c59aba307fb11dc9ba Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 02:40:53 -0700 Subject: [PATCH 10/70] test(services): cover llama.cpp opt-in interop Signed-off-by: Sherief Farouk --- services/tests/llamacpp_interop_test.go | 146 ++++++++++++++++++++++++ services/tests/models_http_test.go | 14 ++- services/tests/models_refresh_test.go | 8 +- 3 files changed, 161 insertions(+), 7 deletions(-) create mode 100644 services/tests/llamacpp_interop_test.go diff --git a/services/tests/llamacpp_interop_test.go b/services/tests/llamacpp_interop_test.go new file mode 100644 index 00000000..4c3c5c58 --- /dev/null +++ b/services/tests/llamacpp_interop_test.go @@ -0,0 +1,146 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package tests + +import ( + "bytes" + "encoding/json" + "fmt" + "io" + "net/http" + "net/http/httptest" + "sync/atomic" + "testing" + "time" +) + +func TestLlamaCPPProxyIsExcludedFromBrokerDefaults(t *testing.T) { + stdin, msgs, stderr, cleanup := startBrokerWith(t, "--proxy-path", proxyBin) + t.Cleanup(cleanup) + go func() { + for range stderr { + } + }() + + waitForMethod(t, msgs, "app:ready", 10*time.Second) + const requestID = 7100 + writeRawFrame(t, stdin, fmt.Sprintf( + `{"jsonrpc":"2.0","id":%d,"method":"llamacpp-proxy:get-status"}`, requestID, + )) + response := waitForResponseID(t, msgs, requestID, 5*time.Second) + if response.Error != nil { + t.Fatalf("llamacpp-proxy:get-status failed: %d %s", response.Error.Code, response.Error.Message) + } + var status struct { + Ready bool `json:"ready"` + Port int `json:"port"` + } + if err := json.Unmarshal(response.Result, &status); err != nil { + t.Fatalf("decode llama.cpp status %s: %v", response.Result, err) + } + if status.Ready || status.Port != 0 { + t.Fatalf("default llama.cpp status = %+v, want disabled", status) + } +} + +func TestLlamaCPPOptInFacadeUsesRouterInventoryAndExactModelIDs(t *testing.T) { + const model = "org/router-model-GGUF:Q4_K_M" + var modelListHits atomic.Int32 + var inferenceHits atomic.Int32 + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + switch { + case r.Method == http.MethodGet && r.URL.Path == "/models": + modelListHits.Add(1) + _ = json.NewEncoder(w).Encode(map[string]any{ + "object": "list", + "data": []map[string]string{{"id": model}}, + }) + case r.Method == http.MethodPost && r.URL.Path == "/v1/chat/completions": + inferenceHits.Add(1) + _, _ = io.Copy(io.Discard, r.Body) + _, _ = io.WriteString(w, `{"choices":[]}`) + default: + http.NotFound(w, r) + } + })) + t.Cleanup(upstream.Close) + + stdin, msgs, stderr, cleanup := startBrokerWith(t, + "--proxy-path", proxyBin, "--proxy-engines", "llamacpp", + ) + t.Cleanup(cleanup) + go func() { + for range stderr { + } + }() + + waitForMethod(t, msgs, "app:ready", 10*time.Second) + proxyPort := waitEngineProxyReady(t, "llamacpp-proxy", stdin, msgs, 15*time.Second) + callBrokerRPC(t, stdin, msgs, 7200, "llamacpp-proxy:node/add-manual", map[string]any{ + "id": "llamacpp-owner", + "host": "127.0.0.1", + "port": portOfURL(t, upstream.URL), + "addresses": []string{"127.0.0.1"}, + "models": []string{model}, + }) + + client := &http.Client{Timeout: 5 * time.Second} + t.Cleanup(client.CloseIdleConnections) + t.Run("remaps OpenAI model list to router inventory", func(t *testing.T) { + response, err := client.Get(fmt.Sprintf("http://127.0.0.1:%d/v1/models", proxyPort)) + if err != nil { + t.Fatalf("get model list: %v", err) + } + defer response.Body.Close() + var list struct { + Object string `json:"object"` + Data []struct { + ID string `json:"id"` + } `json:"data"` + } + if err := json.NewDecoder(response.Body).Decode(&list); err != nil { + t.Fatalf("decode model list: %v", err) + } + found := false + for _, item := range list.Data { + if item.ID == model { + found = true + } + } + if response.StatusCode != http.StatusOK || list.Object != "list" || + !found || modelListHits.Load() != 1 { + t.Fatalf("model list status=%d body=%+v upstreamHits=%d", + response.StatusCode, list, modelListHits.Load()) + } + }) + + post := func(t *testing.T, requestedModel string) int { + t.Helper() + endpoint := fmt.Sprintf("http://127.0.0.1:%d/v1/chat/completions", proxyPort) + response, err := client.Post(endpoint, "application/json", + bytes.NewBufferString(fmt.Sprintf(`{"model":%q,"messages":[]}`, requestedModel))) + if err != nil { + t.Fatalf("post inference: %v", err) + } + if _, err := io.Copy(io.Discard, response.Body); err != nil { + t.Fatalf("read inference response: %v", err) + } + if err := response.Body.Close(); err != nil { + t.Fatalf("close inference response: %v", err) + } + return response.StatusCode + } + t.Run("routes only the exact advertised model id", func(t *testing.T) { + if status := post(t, model); status != http.StatusOK { + t.Fatalf("matching model status = %d, want 200", status) + } + if status := post(t, "org/router-model-GGUF:q4_k_m"); status != http.StatusBadGateway { + t.Fatalf("case-changed model status = %d, want 502", status) + } + if got := inferenceHits.Load(); got != 1 { + t.Fatalf("upstream inference hits = %d, want only the exact match", got) + } + }) +} diff --git a/services/tests/models_http_test.go b/services/tests/models_http_test.go index baf5601d..4a49af10 100644 --- a/services/tests/models_http_test.go +++ b/services/tests/models_http_test.go @@ -33,14 +33,16 @@ func TestModelsHTTPEnrichment(t *testing.T) { // Flat union + per-engine attribution, exactly the shape engine-manager's // ModelsResult serializes. The daemon must enrich both onto the node. _ = json.NewEncoder(w).Encode(map[string]any{ - "models": []string{"llama3:8b", "qwen:0.5b"}, + "models": []string{"llama3:8b", "qwen:0.5b", "org/router-model-GGUF:Q4_K_M"}, "modelsByEngine": map[string][]string{ "ollama": {"llama3:8b"}, "lmstudio": {"qwen:0.5b"}, + "llamacpp": {"org/router-model-GGUF:Q4_K_M"}, }, "loadedByEngine": map[string][]string{ "ollama": {"llama3:8b"}, "lmstudio": {}, + "llamacpp": {"org/router-model-GGUF:Q4_K_M"}, }, }) })) @@ -110,7 +112,7 @@ func findNode(nodes []availableNode, id string) (availableNode, bool) { } func modelsMatch(models []string) bool { - want := []string{"llama3:8b", "qwen:0.5b"} + want := []string{"llama3:8b", "qwen:0.5b", "org/router-model-GGUF:Q4_K_M"} if len(models) != len(want) { return false } @@ -126,17 +128,19 @@ func modelsByEngineMatch(byEngine map[string][]string) bool { want := map[string][]string{ "ollama": {"llama3:8b"}, "lmstudio": {"qwen:0.5b"}, + "llamacpp": {"org/router-model-GGUF:Q4_K_M"}, } return byEngineEqual(byEngine, want) } -// loadedByEngineMatch asserts the loaded set the stub served (ollama has one -// resident model; lmstudio is running but empty) survives the daemon->broker -// projection onto the client-facing node. +// loadedByEngineMatch asserts the loaded set the stub served (Ollama and +// llama.cpp each have one resident model; LM Studio is running but empty) +// survives the daemon->broker projection onto the client-facing node. func loadedByEngineMatch(loaded map[string][]string) bool { want := map[string][]string{ "ollama": {"llama3:8b"}, "lmstudio": {}, + "llamacpp": {"org/router-model-GGUF:Q4_K_M"}, } return byEngineEqual(loaded, want) } diff --git a/services/tests/models_refresh_test.go b/services/tests/models_refresh_test.go index 60878798..d04039ce 100644 --- a/services/tests/models_refresh_test.go +++ b/services/tests/models_refresh_test.go @@ -93,14 +93,16 @@ func TestModelsPeriodicRefreshConvergesWithoutMDNSChange(t *testing.T) { // only the periodic refresh loop can converge the directory. The deadline // comfortably exceeds the refresh interval so at least one sweep runs. stub.set(map[string]any{ - "models": []string{"llama3:8b", "qwen:0.5b"}, + "models": []string{"llama3:8b", "qwen:0.5b", "org/router-model-GGUF:Q4_K_M"}, "modelsByEngine": map[string][]string{ "ollama": {"llama3:8b"}, "lmstudio": {"qwen:0.5b"}, + "llamacpp": {"org/router-model-GGUF:Q4_K_M"}, }, "loadedByEngine": map[string][]string{ "ollama": {"llama3:8b"}, "lmstudio": {}, + "llamacpp": {"org/router-model-GGUF:Q4_K_M"}, }, }) pollForNode(t, stdin, msgs, instance, 45*time.Second, func(n availableNode) bool { @@ -116,13 +118,15 @@ func TestModelsPeriodicRefreshConvergesWithoutMDNSChange(t *testing.T) { "modelsByEngine": map[string][]string{ "ollama": {}, "lmstudio": {}, + "llamacpp": {}, }, "loadedByEngine": map[string][]string{ "ollama": {}, "lmstudio": {}, + "llamacpp": {}, }, }) - emptyByEngine := map[string][]string{"ollama": {}, "lmstudio": {}} + emptyByEngine := map[string][]string{"ollama": {}, "lmstudio": {}, "llamacpp": {}} pollForNode(t, stdin, msgs, instance, 45*time.Second, func(n availableNode) bool { return len(n.Models) == 0 && byEngineEqual(n.ModelsByEngine, emptyByEngine) && byEngineEqual(n.LoadedByEngine, emptyByEngine) From 6dc30a33b4cbaa5cc45dec83d765c6d4ef1c009b Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 13:44:43 -0700 Subject: [PATCH 11/70] docs(services): document opt-in llama.cpp support Signed-off-by: Sherief Farouk --- desktop/docs/services-api.md | 12 ---- desktop/docs/services-parity.md | 46 +++++++++++---- services/nvpair-engine-manager/README.md | 22 ++++++- services/nvpair-engine-manager/spec.md | 30 +++++++++- services/nvpair-proxy/README.md | 32 +++++----- services/nvpair-proxy/spec.md | 29 +++++----- services/nvpair-ui-broker/README.md | 74 +++++++++++++++++++----- services/nvpair-ui-broker/engineproxy.go | 6 +- services/readme.md | 22 ++++--- 9 files changed, 191 insertions(+), 82 deletions(-) diff --git a/desktop/docs/services-api.md b/desktop/docs/services-api.md index 7aea40f9..bc35d1d0 100644 --- a/desktop/docs/services-api.md +++ b/desktop/docs/services-api.md @@ -47,12 +47,8 @@ - ⚠️ nvpair-ui-broker → engine:set-reserved-port - ⚠️ nvpair-ui-broker → engine:unsubscribe - ⚠️ nvpair-ui-broker → internal:set-reserved-port -- ⚠️ nvpair-ui-broker → lmstudio-proxy:get-status - ⚠️ nvpair-ui-broker → lmstudio-proxy:set-port -- ⚠️ nvpair-ui-broker → lmstudio-proxy:unsubscribe -- ⚠️ nvpair-ui-broker → ollama-proxy:get-status - ⚠️ nvpair-ui-broker → ollama-proxy:set-port -- ⚠️ nvpair-ui-broker → ollama-proxy:unsubscribe - ⚠️ nvpair-ui-broker → workloads:unsubscribe ### Backend binaries not listed in `modular-binaries.ts` @@ -253,8 +249,6 @@ | `errors:clear` | notification (we consume) | ✅ yes | | `errors:report` | notification (we consume) | ✅ yes | | `errors:update` | notification (we consume) | ✅ yes | -| `lmstudio-proxy:ready` | notification (we consume) | ➖ ignored | -| `ollama-proxy:ready` | notification (we consume) | ➖ ignored | | `workloads:upsert` | notification (we consume) | ✅ yes | | `connection/cluster-auto-sync` | request (we call) | ➖ ignored | | `connection/cluster-identity` | request (we call) | ✅ yes | @@ -273,20 +267,14 @@ | `engine:unsubscribe` | request (we call) | ⚠️ not called | | `errors:get-initial` | request (we call) | ✅ yes | | `internal:set-reserved-port` | request (we call) | ⚠️ not called | -| `lmstudio-proxy:get-status` | request (we call) | ⚠️ not called | | `lmstudio-proxy:set-port` | request (we call) | ⚠️ not called | -| `lmstudio-proxy:subscribe` | request (we call) | ✅ yes | -| `lmstudio-proxy:unsubscribe` | request (we call) | ⚠️ not called | | `node/add` | request (we call) | ✅ yes | | `node/discovered` | request (we call) | ✅ yes | | `node/remove` | request (we call) | ✅ yes | | `node/removed` | request (we call) | ✅ yes | | `node/updated` | request (we call) | ✅ yes | | `nodes/list` | request (we call) | ✅ yes | -| `ollama-proxy:get-status` | request (we call) | ⚠️ not called | | `ollama-proxy:set-port` | request (we call) | ⚠️ not called | -| `ollama-proxy:subscribe` | request (we call) | ✅ yes | -| `ollama-proxy:unsubscribe` | request (we call) | ⚠️ not called | | `ready` | request (we call) | ✅ yes | | `workloads:get-initial` | request (we call) | ✅ yes | | `workloads:remove` | request (we call) | ✅ yes | diff --git a/desktop/docs/services-parity.md b/desktop/docs/services-parity.md index c0b8cbc5..796decad 100644 --- a/desktop/docs/services-parity.md +++ b/desktop/docs/services-parity.md @@ -24,6 +24,7 @@ history. | Manual nodes | Complete with local persistence | Broker owns probing and proxy registration; Electron persists entries for replay | | Ollama routing | Complete | Broker relay and backend scheduler drive proxy routing | | LM Studio routing | Complete | Parallel broker relay and scheduler path | +| llama.cpp backend | Backend-only opt-in | Services can manage and route it; Electron and renderer contracts do not expose it yet | | Local engine lifecycle | Complete | Install, start, stop, uninstall, update, and port configuration | | Remote engine lifecycle | Partial | Remote install, start, stop, status, and model pull are supported | | Engine models | Partial | Core list, pull, load, unload, and supported delete actions are wired | @@ -99,26 +100,28 @@ they survive worker restarts. ## Routing and inference -Both text-engine facades are broker-owned and cluster-aware. They live in one +All engine facades are broker-owned and cluster-aware. They live in one `nvpair-proxy` process, each enabled after spawn on its own port, and each serves its engine's dialect: - the Ollama facade serves the Ollama-compatible surface; -- the LM Studio facade serves the LM Studio/OpenAI-compatible surface. +- the LM Studio facade serves the LM Studio/OpenAI-compatible surface; +- the opt-in llama.cpp facade serves OpenAI-compatible routes and remaps + `GET /v1/models` to the router's `GET /models`. Sharing a process is what lets them share the burst reservations the scheduler -depends on: two facades bursting at once compete for the same node's GPU, so a -dispatch through either has to be visible to the other. +depends on: facades bursting at once compete for the same node's GPU, so a +dispatch through any one has to be visible to the others. Routing precedence is manual selection, scheduler priority, then deterministic proxy ordering. Personal AI Router leaves proxies in automatic mode. -`nvpair-job-scheduler` combines total queued and running workload across both -engines with a smoothed 0–3 GPU-pressure signal. The backend scanner and manual -node worker provide maximum-GPU utilization, while invalid, missing, or -older-than-10-second samples receive neutral pressure. The scheduler emits order, -pending count, and pressure; the broker forwards each `schedule:priority` -snapshot to the matching proxy through `node/set-priority`. Each proxy adds +`nvpair-job-scheduler` combines total queued and running workload across all +enabled engines with a smoothed 0–3 GPU-pressure signal. The backend scanner and +manual node worker provide maximum-GPU utilization, while invalid, missing, or +older-than-10-second samples receive neutral pressure. The scheduler emits +order, pending count, and pressure; the broker forwards each `schedule:priority` +snapshot to the matching facade through `node/set-priority`. Each facade adds local reservations, so its estimate is `pending + gpuPressure + localReservations` during concurrent bursts. @@ -133,8 +136,8 @@ LAN-reachable. Each node's proxy exposes two personalities on one listener: a loopback-only plaintext path for local clients, and a LAN ingress gated by cluster mTLS that forwards trusted-peer requests to the loopback engine. Because the engine port is private, discovery advertises the **promoted proxy port** for -`ol`/`lm`, and the peer's real engine port is knowable only from authoritative -`engine:remote-get-installed` facts. +`ol`/`lm`/`lc`, and the peer's real engine port is knowable only from +authoritative `engine:remote-get-installed` facts. Personal AI Router consequences (all reflection, no security implementation): @@ -145,6 +148,25 @@ Personal AI Router consequences (all reflection, no security implementation): run inference across the version boundary. Local use and the shared nearby-model list are unaffected. +### llama.cpp backend checkpoint + +The bundled backend manifest can install and start `llama-server`, list exact +router model ids, stream model downloads over SSE, and load or unload a model. +It declares no delete action, so persistent downloads currently require manual +cache cleanup. Its `LLAMA_CACHE` directory is a sibling of the install directory +and survives uninstall and reinstall. + +Windows and Linux installs use checksum-pinned server and CUDA-runtime archive +pairs; macOS uses the standard Metal-capable archive. The on-demand download is +roughly 0.6–0.8 GiB and is not part of the application installer. GPU layers +remain `auto`, allowing supported NVIDIA/Metal acceleration and dynamic CPU +fallback; hardware acceptance is still required to confirm acceleration. + +The broker keeps the facade out of its default set. A backend operator must pass +`--proxy-engines ollama,lmstudio,llamacpp`, which places the facade on `8080` +and the managed router on `8081`. Electron has no llama.cpp engine identity, +catalog, bridge mapping, or renderer workflow at this checkpoint. + ## Engine lifecycle `nvpair-engine-manager` is authoritative for installed, running, healthy, and diff --git a/services/nvpair-engine-manager/README.md b/services/nvpair-engine-manager/README.md index b401c883..2d859c15 100644 --- a/services/nvpair-engine-manager/README.md +++ b/services/nvpair-engine-manager/README.md @@ -5,15 +5,31 @@ SPDX-License-Identifier: Apache-2.0 # nvpair-engine-manager -A config-driven control plane for local inference engines (Ollama today; -Intel/others via a dropped-in manifest). It manages everything about an -engine **except serving inference**: detect, user-mode install, +A config-driven control plane for local inference engines. The bundled +manifests support Ollama, LM Studio, and llama.cpp. It manages everything about +an engine **except serving inference**: detect, user-mode install, start/stop/restart, health, and config-declared actions. Adding an engine is a JSON manifest, not code. The bundled manifests under `manifests/` are the working reference for manifest authoring. +### llama.cpp backend checkpoint + +The `llamacpp` manifest runs `llama-server` in router mode on loopback port +`8081`. It can list, download, load, and unload exact model ids such as +`owner/repository:Q4_K_M`; deletion is not declared. Downloads use `/models/sse` +for progress, and `LLAMA_CACHE` points to a managed sibling directory so models +survive engine uninstall and reinstall. Remove that cache manually when needed. + +Windows and Linux installs download checksum-pinned server and CUDA-runtime +archive pairs (CUDA 12.x for x64 and CUDA 13.4 for arm64); macOS uses the +standard Metal-capable archive. These on-demand downloads are roughly +0.6–0.8 GiB and do not enlarge the PAIR installer. llama.cpp keeps its `auto` +GPU-layer policy: supported NVIDIA/Metal devices can accelerate, while the +dynamic CPU backend remains the fallback. Hardware acceptance, not `/health` +alone, is required to claim GPU activation. + ## Communication Bidirectional newline-delimited JSON-RPC 2.0 — the same conventions as diff --git a/services/nvpair-engine-manager/spec.md b/services/nvpair-engine-manager/spec.md index ba2e6324..8cdc831a 100644 --- a/services/nvpair-engine-manager/spec.md +++ b/services/nvpair-engine-manager/spec.md @@ -6,7 +6,13 @@ SPDX-License-Identifier: Apache-2.0 # Microservice: Engine Manager (`nvpair-engine-manager`) ## 1. Purpose -A declarative, config-driven control plane for **local inference engines** (Ollama today; Intel / llama.cpp / others later). It owns an engine's entire lifecycle *except serving inference* — locate, install, launch, stop, restart, health, and config-declared actions — so one uniform API manages any engine across OSes with no per-engine code. A third party drops in a JSON manifest and their engine's install/launch/controls "just appear" over the same API: the core extensibility story for an open-source product. +A declarative, config-driven control plane for **local inference engines** +(currently Ollama, LM Studio, and llama.cpp). It owns an engine's entire +lifecycle *except serving inference* — locate, install, launch, stop, restart, +health, and config-declared actions — so one uniform API manages any engine +across OSes with no per-engine code. A third party drops in a JSON manifest and +their engine's install/launch/controls "just appear" over the same API: the core +extensibility story for an open-source product. ## 2. Scope **In scope** @@ -22,7 +28,19 @@ A declarative, config-driven control plane for **local inference engines** (Olla host-platform port overrides, deleting the file only when it has no other settings. Host-platform precedence must not override a successfully saved port on reload; malformed existing overrides fail the save and remain intact. -- Config-declared **actions** covering the full model lifecycle — Ollama: `list_models`, `loaded_models`, `pull_model`, `run_model`, `unload_model`, `delete_model`; LM Studio: `list_models`/`list_downloaded`, `loaded_models`, `pull_model`, `load_model`, `chat`, `unload_model`, `delete_model` (`remove_path` with `lms-disk-path` resolution) — mapped to each engine's local control API. `loaded_models` reports the models currently resident in memory (Ollama `GET /api/ps`, LM Studio `GET /api/v1/models` filtered by nonempty `loaded_instances`), name-extracted via the same declarative `result` spec (with an optional `match` row filter). +- Config-declared **actions** mapped to each engine's local control API: + Ollama declares `list_models`, `loaded_models`, `pull_model`, `run_model`, + `unload_model`, and `delete_model`; LM Studio declares + `list_models`/`list_downloaded`, `loaded_models`, `pull_model`, `load_model`, + `chat`, `unload_model`, and `delete_model` (`remove_path` with + `lms-disk-path` resolution); + llama.cpp declares list, loaded-list, pull, load, and unload with exact model + ids, and intentionally declares no delete action. `loaded_models` reports + models currently resident in memory, name-extracted via the same declarative + `result` spec with optional nested-path and row filters: Ollama uses + `GET /api/ps`, LM Studio filters nonempty `loaded_instances` from + `GET /api/v1/models`, and llama.cpp matches nested `status.value == loaded` + records from `GET /models`. - Per-engine stdout/stderr log capture and structured operational error records, surfaced via the errors pipeline. - A normalized node-level model list (`engine:models`): union of every running engine's `list_models`, name-extracted via each action's declarative `result` spec, plus the per-engine set of models loaded in memory (`loadedByEngine`, from each engine's `loaded_models` action). A successful explicit empty inventory remains an engine key with `[]`; a missing/malformed/failed inventory omits that engine key instead of being mislabeled as authoritative empty. A watcher polls the loaded set and pushes `engine:models-changed` when it changes (explicit load/unload, JIT auto-load, TTL/idle eviction). - Expose all of the above over the `engine:*` JSON-RPC surface to whatever orchestrates the service, plus an optional plain-HTTP LAN endpoint (`--http-port`, `GET /v1/models`) that serves the model list to a peer's discovery daemon (the list moved off the size-limited mDNS TXT onto HTTP). @@ -31,6 +49,7 @@ A declarative, config-driven control plane for **local inference engines** (Olla - **Inference traffic** — stays with `nvpair-proxy`; this service never proxies `/api/chat` etc. - **Multi-instance per engine and an MCP server** — future-additive, not v1. - **The node's error list** — owned by `nvpair-errors`, which holds it as in-memory session state; this service only emits `errors:report` / `errors:clear`. +- Automatic cleanup of persistent llama.cpp model downloads. ## 3. Key Use Cases - **Install an engine, user-mode**: `engine:install {engine:"ollama"}` downloads the per-OS user-scoped package (Windows/Linux standalone archive extracted into a user dir; macOS app bundle — never an elevated `Setup.exe` or `curl | sh`), checksum-verifies, extracts, re-detects. @@ -160,7 +179,12 @@ The `engine:remote-*` methods are the client half: engine-manager resolves the t ## 9. Data Ownership - **Owned**: the in-memory engine registry (parsed manifests + per-engine runtime state) and per-engine log/error ring buffers — transient only. - **Source of truth**: no — `nvpair-errors` owns the node's error list (in memory, for the session); model inventories belong to the engines; manifests on disk are authored elsewhere. -- **Storage**: in-memory; manifests read from the per-user data dir's `engines/*.json` (`%LocalAppData%\Nvidia Corporation\Personal AI Router` on Windows, `~/.config/Nvidia Corporation/Personal AI Router` on Linux, `~/Library/Application Support/Nvidia Corporation/Personal AI Router` on macOS) plus bundled `manifests/*.json`. No database. +- **Storage**: in-memory; manifests read from the per-user data dir's + `engines/*.json` (`%LocalAppData%\Nvidia Corporation\Personal AI Router` on + Windows, `~/.config/Nvidia Corporation/Personal AI Router` on Linux, and + `~/Library/Application Support/Nvidia Corporation/Personal AI Router` on + macOS) plus bundled `manifests/*.json`. llama.cpp uses a managed sibling cache + that survives uninstall and must currently be removed manually. No database. ## 10. Design Constraints - **Performance**: control plane, not inference; sub-second RPCs except install (network-bound) and start (bounded by the readiness timeout). diff --git a/services/nvpair-proxy/README.md b/services/nvpair-proxy/README.md index 20c753e1..77b2f85f 100644 --- a/services/nvpair-proxy/README.md +++ b/services/nvpair-proxy/README.md @@ -73,7 +73,7 @@ parameter, because one flag cannot carry two engines' plans. | Field | Default | Description | |-------|---------|-------------| -| `engine` | *(required)* | Which engine to front: `ollama` or `lmstudio`. An unknown name is rejected with the accepted values. | +| `engine` | *(required)* | Which engine to front: `ollama`, `lmstudio`, or `llamacpp`. An unknown name is rejected with the accepted values. | | `port` | per engine, see below | HTTP listen port for request forwarding. Must be 1–65535, or omitted for the engine's standalone default. `0` means "the default" rather than "pick an ephemeral port", and any other out-of-range value is rejected, because the facade announces the requested port in its `ready` notification and the broker would be told `0`. | | `aliasAddresses` | *(empty)* | Optional secondary `host:port` values for the same routing handler, one per loopback family so `localhost` resolves either way. Only literal loopback addresses are accepted; the broker uses this for a safe inherited local `OLLAMA_HOST`, and the aliases are not advertised to peers. Accepted only for an engine with an inherited host variable — today Ollama alone — and rejected for any other. | | `ignorePersistedPort` | `false` | Use `port` even when a saved port exists (used by broker-managed startup) | @@ -83,10 +83,10 @@ parameter, because one flag cannot carry two engines' plans. Everything engine-specific is one entry in `engines.go`, plus the shared identity in `nvpair-shared/engines`. -| | `"engine":"ollama"` | `"engine":"lmstudio"` | -|---|---|---| -| Facade id — error-ID prefix, broker relay namespace, TUI proxies-view tab | `ollama-proxy` | `lmstudio-proxy` | -| Discovery service key | `ol` | `lm` | +| | `"engine":"ollama"` | `"engine":"lmstudio"` | `"engine":"llamacpp"` | +|---|---|---|---| +| Facade id — error-ID prefix and broker relay namespace | `ollama-proxy` | `lmstudio-proxy` | `llamacpp-proxy` | +| Discovery service key | `ol` | `lm` | `lc` | The **log component, supervisor label, and TUI health crash key are not in that table**: they name the process (`nvpair-proxy`), not a facade, because one @@ -95,20 +95,24 @@ Facade-scoped log records carry an `engine` field instead. The supervisor label and the health crash key are matched against each other, so they move together — see `nvpair-shared/engines` and `spec.md` §9. -| | `"engine":"ollama"` | `"engine":"lmstudio"` | -|---|---|---| -| Engine's own client-facing port | 11434 | 1234 | -| Where PAIR relocates the engine | 11435 | 1235 | -| Standalone port, used when `port` is omitted | 11435 | 1234 | -| Persisted-port file (declared, not derived) | `proxy-port.json` | `lmstudio-proxy-port.json` | -| Model-list routes | `GET /api/tags` (native), `GET /v1/models` (OpenAI) | `GET /v1/models` (OpenAI) | -| Inference routes | `/api/generate`, `/api/chat`, `/api/embeddings`, `/api/embed`, plus the OpenAI and Anthropic Messages sets | `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings`, `/v1/messages` | -| Model naming | untagged means `:latest`, so `llama3` and `llama3:latest` are one model | identifiers compared byte for byte | +| | `"engine":"ollama"` | `"engine":"lmstudio"` | `"engine":"llamacpp"` | +|---|---|---|---| +| Engine's own client-facing port | 11434 | 1234 | 8080 | +| Where PAIR relocates the engine | 11435 | 1235 | 8081 | +| Standalone port, used when `port` is omitted | 11435 | 1234 | 8080 | +| Persisted-port file (declared, not derived) | `proxy-port.json` | `lmstudio-proxy-port.json` | `llamacpp-proxy-port.json` | +| Model-list routes | `GET /api/tags` (native), `GET /v1/models` (OpenAI) | `GET /v1/models` (OpenAI) | `GET /v1/models` (OpenAI; upstream `/models`) | +| Inference routes | `/api/generate`, `/api/chat`, `/api/embeddings`, `/api/embed`, plus the OpenAI and Anthropic Messages sets | `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings`, `/v1/messages` | `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings` | +| Model naming | untagged means `:latest`, so `llama3` and `llama3:latest` are one model | identifiers compared byte for byte | identifiers compared byte for byte | The route table is a **classifier, not an allowlist**. An unlisted path is forwarded verbatim, which is how `/api/show`, `/api/pull`, `/api/ps`, `/api/version` and `OPTIONS` preflights keep working. +The broker enables only Ollama and LM Studio by default. Opt in to the +llama.cpp facade with `--proxy-engines llamacpp`; local OpenAI-compatible +clients then use `8080`, while the managed `llama-server` stays on `8081`. + ### HTTP Reverse Proxy A facade listens on its enabled port and forwards incoming requests to the diff --git a/services/nvpair-proxy/spec.md b/services/nvpair-proxy/spec.md index 18c59999..cd3e8a77 100644 --- a/services/nvpair-proxy/spec.md +++ b/services/nvpair-proxy/spec.md @@ -98,17 +98,17 @@ The process starts with **no engine and no listener**. The broker then sends one `facade/enable` per engine, carrying that engine's port and any alias addresses. A flag cannot express this. The broker plans a different port for each engine — -Ollama's managed facade wants `:11434` while LM Studio's wants `:1234`, and -either may be absent so the child keeps its own persisted port — and a -single-valued flag carries only one plan. +Ollama's managed facade wants `:11434`, LM Studio's wants `:1234`, and +llama.cpp's opt-in facade wants `:8080`; any may be absent so the child keeps +its own persisted port — and a single-valued flag carries only one plan. ### 3.1 Why one process Between scheduler snapshots a facade takes short-lived **reservations** for work it has dispatched but that the scheduler has not yet observed. Those live in the process, and every facade shares them. If each engine had its own proxy -process, neither could see the other's reservations, so simultaneous bursts of -Ollama and LM Studio requests could both select the same node, each incorrectly +process, none could see the others' reservations, so simultaneous bursts across +Ollama, LM Studio, and llama.cpp could select the same node, each incorrectly believing it was idle. Sharing the map is the reason the engines share a process. @@ -163,8 +163,8 @@ For a model-bearing inference request: 1. Filter a request-local discovery snapshot to nodes whose per-engine inventory advertises the requested model. Ollama normalizes the implicit `:latest` tag; - LM Studio ids match exactly. An empty owner set returns a local `502` without - contacting an engine. + LM Studio and llama.cpp ids match exactly. An empty owner set returns a local + `502` without contacting an engine. 2. Order the eligible owners: explicit `node/select` pin, then the scheduler's priority list, then deterministic default ordering. 3. Reserve the least estimated-loaded scheduler-listed candidate and move it to @@ -178,6 +178,9 @@ For a model-bearing inference request: An ineligible manual selection cannot override the capability gate, and failover never broadens to an excluded node. +The llama.cpp facade exposes the OpenAI-compatible `GET /v1/models` route and +remaps it to the router's `GET /models`; its inference routes remain `/v1/*`. + ### 5.1 Retry bounds | Bound | Value | Governs | @@ -549,12 +552,12 @@ untouched, the primary listener stays up, and a warning is reported. ## 8. Ports -| | Ollama | LM Studio | -| --- | --- | --- | -| Engine's own client-facing port | 11434 | 1234 | -| Where PAIR relocates the engine | 11435 | 1235 | -| Standalone port, when `port` is omitted | 11435 | 1234 | -| Persisted-port file | `proxy-port.json` | `lmstudio-proxy-port.json` | +| | Ollama | LM Studio | llama.cpp | +| --- | --- | --- | --- | +| Engine's own client-facing port | 11434 | 1234 | 8080 | +| Where PAIR relocates the engine | 11435 | 1235 | 8081 | +| Standalone port, when `port` is omitted | 11435 | 1234 | 8080 | +| Persisted-port file | `proxy-port.json` | `lmstudio-proxy-port.json` | `llamacpp-proxy-port.json` | A port chosen at runtime via `set-port` is persisted per engine and restored when that facade is enabled, taking precedence over the requested port, so the diff --git a/services/nvpair-ui-broker/README.md b/services/nvpair-ui-broker/README.md index 91cd7f91..4622505e 100644 --- a/services/nvpair-ui-broker/README.md +++ b/services/nvpair-ui-broker/README.md @@ -23,7 +23,7 @@ namespace: | --- | --- | --- | | `nvpair-node-scanner` | Discovery daemon: advertises this host's one `_nvpair-node._tcp` record and browses the LAN | `discovery:*` | | `nvpair-node-info` | Local GPU / CPU / memory inventory over HTTP at `/v1/node-info` | — (HTTP only) | -| `nvpair-proxy` | One process hosting an inference proxy and router facade per enabled engine | `ollama-proxy:*`, `lmstudio-proxy:*` | +| `nvpair-proxy` | One process hosting an inference proxy and router facade per enabled engine | `ollama-proxy:*`, `lmstudio-proxy:*`, `llamacpp-proxy:*` | | `nvpair-engine-manager` | Local engine and model control plane; also serves `GET /v1/models` to peers | `engine:*` | | `nvpair-cluster-manager` | Node identity, trusted-node store, PIN pairing | `cluster:*`, `nodes:*` | | `nvpair-workload-manager` | Cluster workload relay between this node and peers | `workloads:*` | @@ -39,9 +39,10 @@ lifecycle, and relay rules. Two responsibilities live in the broker itself rather than in a worker: -- **Engine advertising.** The broker polls local Ollama and LM Studio every 5 s - and registers each running engine's port (`ol` / `lm`) with the discovery - daemon, so both are carried in this host's single `_nvpair-node` record. The +- **Engine advertising.** The broker polls local Ollama, LM Studio, and + llama.cpp every 5 s and registers each running engine's promoted facade port + (`ol` / `lm` / `lc`) with the discovery daemon, so they are carried in this + host's single `_nvpair-node` record. The model list is not part of that record — it is served over HTTP by `nvpair-engine-manager` on the `em` service and fetched by a peer's daemon during discovery enrichment. @@ -73,7 +74,7 @@ Bidirectional newline-delimited JSON-RPC 2.0 — same conventions as every other | `--scanner-path ` | `./nvpair-node-scanner[.exe]` in the CWD | Explicit path to the `nvpair-node-scanner` binary the broker should spawn | | `--node-info-path ` | `./nvpair-node-info[.exe]` in the CWD | Explicit path to the `nvpair-node-info` binary the broker should spawn. When omitted and no default sibling exists, the broker runs without the local inventory server (non-fatal); when set to an invalid path, the broker exits with an error | | `--proxy-path ` | `./nvpair-proxy[.exe]` in the CWD | Explicit path to the `nvpair-proxy` binary. One process fronts every engine: the broker spawns it once and then sends a `facade/enable` per entry in `--proxy-engines`. Same optional semantics as `--node-info-path`: an absent default sibling means no local proxies (non-fatal); an invalid explicit path exits with an error | -| `--proxy-engines ` | every engine in `nvpair-shared/engines` (currently `ollama,lmstudio`) | Which engines to front with a proxy. An unrecognized name exits with an error rather than being skipped, so a typo cannot look like it worked. An engine left out is not started **and not prepared** — the broker will not relocate an engine whose facade nothing is going to claim | +| `--proxy-engines ` | `ollama,lmstudio` | Which engines to front with a proxy. Add `llamacpp` explicitly to opt in. An unrecognized name exits with an error rather than being skipped, so a typo cannot look like it worked. An engine left out is not started **and not prepared** — the broker will not relocate an engine whose facade nothing is going to claim | | `--workload-manager-path ` | `./nvpair-workload-manager[.exe]` in the CWD | Explicit path to the `nvpair-workload-manager` binary the broker spawns for the cluster workload relay. Same optional semantics as `--node-info-path`: an absent default sibling means no workload relay (non-fatal); an invalid explicit path exits with an error | | `--errors-path ` | `./nvpair-errors[.exe]` in the CWD | Explicit path to the `nvpair-errors` binary the broker spawns (with `--peer-sync`) for the service-error pipeline. Same optional semantics as `--node-info-path`: an absent default sibling means the error pipeline is disabled — producers' errors are dropped (non-fatal); an invalid explicit path exits with an error | | `--engine-manager-path ` | `./nvpair-engine-manager[.exe]` in the CWD | Explicit path to the `nvpair-engine-manager` binary the broker spawns for engine management. Same optional semantics as `--node-info-path` | @@ -91,43 +92,78 @@ Logs go to **stderr** (shared `applog` format, same as every other NVPAIR binary On startup — **before** emitting `app:ready` — the broker spawns the scanner and (when available) node-info, `nvpair-proxy`, the workload-manager, and the cluster-manager as child processes over stdio. The proxy is spawned up front but doesn't gate `app:ready` — each of its facades announces its listen port asynchronously (see below). None of the auxiliary workers gate `app:ready`. -**`nvpair-node-scanner`** (the consolidated discovery daemon) is spawned first. It pushes `discovery:node-discovered`, `discovery:node-updated`, and `discovery:node-removed` notifications into the broker, which maintains them in an in-memory map keyed by `id`. Clients query that map via `discovery:get-nodes` and — once they've opted in via `discovery:subscribe` — receive a `discovery:nodes-changed` notification on every store mutation. The raw `discovery:node-*` notifications are never forwarded as-is. The scanner polls healthy node-info endpoints on a staggered two-second cadence, backs consecutive remote failures off to a 30-second cap, and emits compact `discovery:node-telemetry` observations containing maximum GPU utilization, validity, and age; these remain internal to broker scheduling. The broker registers this node's local service ports (`ni`/`er`/`wl`/`cl`/`em`, plus `ol`/`lm` from the engine poller) with the daemon over the same link, so the daemon can advertise them all in one `_nvpair-node` record. +**`nvpair-node-scanner`** (the consolidated discovery daemon) is spawned first. It pushes `discovery:node-discovered`, `discovery:node-updated`, and `discovery:node-removed` notifications into the broker, which maintains them in an in-memory map keyed by `id`. Clients query that map via `discovery:get-nodes` and — once they've opted in via `discovery:subscribe` — receive a `discovery:nodes-changed` notification on every store mutation. The raw `discovery:node-*` notifications are never forwarded as-is. The scanner polls healthy node-info endpoints on a staggered two-second cadence, backs consecutive remote failures off to a 30-second cap, and emits compact `discovery:node-telemetry` observations containing maximum GPU utilization, validity, and age; these remain internal to broker scheduling. The broker registers this node's local service ports (`ni`/`er`/`wl`/`cl`/`em`, plus `ol`/`lm`/`lc` from the engine poller) with the daemon over the same link, so the daemon can advertise them all in one `_nvpair-node` record. **`nvpair-node-info`** is spawned next. It's a server, not an event source: it stands up the local `/v1/node-info` HTTP endpoint (GPU/CPU/memory inventory). It does not advertise itself — the broker registers its `ni` port with the scanner daemon, which carries it in the node record, and a peer's daemon fetches `/v1/node-info` over plain HTTP to enrich the node. The broker doesn't read anything back from node-info's stdout (drained and discarded). Spawning it is **optional**: if the binary can't be resolved (and no `--node-info-path` override was given) the broker logs a warning and continues serving discovery without it. -**Engine advertising.** The broker runs an internal 5 s poll loop against local Ollama at its configured backend port and LM Studio (`GET /v1/models`) and reconciles this node's engine registration with the scanner daemon: +**Engine advertising.** The broker runs an internal 5 s poll loop against local +Ollama (`GET /`), LM Studio (`GET /v1/models`), and llama.cpp (`GET /health`) +at their configured backend ports and reconciles this node's engine registration +with the scanner daemon: -- engine **up** → register `ol` / `lm` at the engine's real port, never the proxy's own, to prevent a self-forward loop; +- engine **up** and facade **ready** → register `ol` / `lm` / `lc` at the + promoted facade port, never the private backend port; - engine **down** → unregister it. The daemon folds those registrations into this host's single `_nvpair-node` record, so a peer discovers the engine through the shared channel. The model list is not part of that registration — it's served over HTTP by `nvpair-engine-manager` (the `em` service, `GET /v1/models`) and enriched onto each node by the peer's daemon. There is no separate advertiser subprocess and no manual-advertise RPC. **`nvpair-proxy`** is one process that fronts every enabled engine. It starts with no engine and no listener; the broker then sends it a `facade/enable` per engine, carrying that engine's port and any alias addresses. A flag could not express this, because the broker plans a different port for each engine. Each facade forwards inference to a node it discovers on the network and speaks its own engine's dialect. -One process for all of them is deliberate. Between scheduler snapshots a facade takes short-lived reservations for work it has dispatched, and those live in the process — two processes each held half that picture, so simultaneous Ollama and LM Studio bursts could both pick the same node believing it idle. The cost is **shared fate**: a crash takes every facade down and the supervisor brings them all back together, reported once as `supervisor:subprocess-crashed:nvpair-proxy` rather than against one engine. Within the process the boundaries are finer — a facade that loses its bind race, cannot be moved to a free port, or panics while handling a request is withdrawn or answered with an error on its own, leaving the others serving. +One process for all of them is deliberate. Between scheduler snapshots a facade +takes short-lived reservations for work it has dispatched, and those live in +the process; separate processes would each hold only part of that picture, so +simultaneous bursts across engines could select the same node believing it idle. +The cost is **shared fate**: a crash takes every facade down and the supervisor +brings them all back together, reported once as +`supervisor:subprocess-crashed:nvpair-proxy` rather than against one engine. +Within the process the boundaries are finer — a facade that loses its bind +race, cannot be moved to a free port, or panics while handling a request is +withdrawn or answered with an error on its own, leaving the others serving. Ollama's standalone default is `:11435`; with managed port ownership enabled (the default), the broker starts settings and engine-manager first, claims `:11434` with that facade, and only then moves a stopped default-port Ollama backend to a free port. Custom backend ports are preserved. When the inherited `OLLAMA_HOST` names a distinct local plaintext port, the broker also gives the facade that normalized loopback-only alias so clients already using the variable enter the same routing path; `localhost` reserves both canonical loopback families atomically, while remote and HTTPS targets are ignored. The alias port is reserved against every configured engine, local or remote engine start override, every facade's control plane, and the managed Ollama and LM Studio backend port plans, so a backend that has to move can never land on the alias. A running Ollama or unknown owner on either requested port is never stopped or moved: the primary uses a safe fallback when needed, and an occupied alias remains with its owner while the broker reports a warning. +llama.cpp is prepositioned by its manifest on `:8081`; the broker never takes +over or moves that process. When explicitly enabled, it places the facade on +`:8080` or a safe fallback without colliding with the fixed backend port. + For automatic model-bearing inference, every facade combines scheduler pending counts and GPU pressure with the process-wide reservation map under one lock before forwarding, so concurrent requests distribute without an artificial delay or a round trip through the scheduler. A reservation is released when its request ends and moves with a failover, so a node stops counting as loaded as soon as it stops working. Manual pins, model-owner tiers, and the complete failover list keep their existing precedence. The broker otherwise treats the proxy as **optional and non-fatal**. Each facade announces its bound port **asynchronously**, via an engine-addressed `ready` notification emitted once its HTTP listener is up. The broker records readiness per engine and exposes it through that engine's `-proxy:get-status` request — per engine, because the facades bind different ports and a single port for the process would be whichever readied last. Because `ready` arrives after `app:ready` (and the proxy is optional), clients learn a port by **polling** `-proxy:get-status` rather than assuming it from `app:ready`. -The proxy is a full bidirectional JSON-RPC peer with a control plane (node selection, manual nodes, ...) and an event stream. The broker acts as a **generic relay** in both directions: any request a client sends under the `ollama-proxy:` namespace (other than the reserved broker-local ones) is forwarded to that engine's facade and the response relayed straight back (see `ollama-proxy:` below), and every notification the facade emits is re-emitted to subscribed clients as `ollama-proxy:` (see `ollama-proxy:` below). +The proxy is a full bidirectional JSON-RPC peer with a control plane (node +selection, manual nodes, ...) and an event stream. The broker acts as a +**generic relay** in both directions: any request a client sends under an +enabled `-proxy:` namespace (other than reserved broker-local ones) is +forwarded to that engine's facade and the response relayed straight back, and +every facade notification is re-emitted to subscribed clients under the same +component namespace. Client namespaces are unchanged by the process collapse, but the wire inside is not. Because one process holds every facade, a message on that link carries the engine it concerns: the client's `ollama-proxy:` prefix comes off and a bare `ollama:` facade address goes on. The two are not interchangeable — `ollama-proxy:` is how a client addresses the component, `ollama:` is how a message addresses a facade inside the process — so the relay is a translation between them rather than a strip. Process-scoped methods (`log/set-level`, the scheduler's `node/set-priority`) carry no address, and `facade/enable` names its engine in the payload because it runs before that facade exists. -Two classes of proxy notification are **not** re-emitted under the `ollama-proxy:` namespace, because neither is a proxy control-plane event: +Two classes of proxy notification are **not** re-emitted under any +`-proxy:` namespace, because neither is a proxy control-plane event: - The `workload:*` lifecycle events the proxy fires per inference request. Those are workload-manager traffic — see the workload-manager paragraph below for how they're routed. - `node/activity`, which a proxy raises while a peer's engine is streaming response bytes back through it. That is discovery input: the broker forwards it to `nvpair-node-scanner` as a `discovery:node-activity` notification, where it counts as proof the peer is alive and cancels the eviction it would otherwise face for failing a liveness probe it had no spare CPU to answer. No client has any use for a per-request liveness frame. The handoff is a bounded queue drained by one goroutine — reports arrive for as long as inference streams, so a wedged scanner must not be able to stall the proxy reader, and a dropped report only means the scanner falls back to probing a node that will very likely answer. **`nvpair-workload-manager`** is the cluster workload relay, and it's the only worker the broker talks to **bidirectionally over a notification-only link** (no id-bearing request/response). It supervises it the same optional, non-fatal way as node-info / proxy: a missing default sibling (and no `--workload-manager-path`) just means no cluster workload relay. The broker plays the workload **broker** role between the proxy and the manager: -- **Outbound (proxy -> broker -> manager -> peers).** When either proxy emits a `workload:started` / `workload:completed` / `workload:errored`, the broker stamps the local stable `hostUuid` onto `params.workloadInfo.originatedFrom`, applies the transition to its authoritative workload store, and fans the accepted update to the scheduler before forwarding the original lifecycle frame to the manager. The manager broadcasts it to peer nodes. With no manager supervised the event still updates local scheduling and subscribed clients, but is not broadcast. +- **Outbound (proxy -> broker -> manager -> peers).** When a facade emits a `workload:started` / `workload:completed` / `workload:errored`, the broker stamps the local stable `hostUuid` onto `params.workloadInfo.originatedFrom`, applies the transition to its authoritative workload store, and fans the accepted update to the scheduler before forwarding the original lifecycle frame to the manager. The manager broadcasts it to peer nodes. With no manager supervised the event still updates local scheduling and subscribed clients, but is not broadcast. - **Inbound (peers -> manager -> broker).** The manager translates peer-origin lifecycle events into `workloads:upsert` and peer-origin removals into `workloads:remove` on stdout. The broker applies each accepted transition to the same store, fans it to the scheduler, and relays it to clients subscribed via `workloads:subscribe`. - **Local echo.** Local-origin proxy workloads are also emitted to the same `workloads:*` client stream (lifecycle translated to `workloads:upsert`), so a subscribed client sees a coherent cluster-wide view — its own workloads alongside peers'. -**`nvpair-job-scheduler`** consumes the accepted workload stream, compact GPU telemetry, and discovery snapshot. It smooths fresh utilization into pressure 0–3, uses neutral pressure 1 for invalid/missing/older-than-10-second samples, and orders by `pending + gpuPressure`, then pressure, then stable UUID. Load is node-wide across Ollama and LM Studio because both normally contend for the same resources. Each engine-specific `schedule:priority` carries `{engine,nodes,ranks}` and refreshes when order, pending counts, or pressure changes. The broker caches, generation-orders, and replays the full `{nodes,ranks}` snapshot to the matching proxy, where a newly delivered snapshot resets optimistic reservation deltas. On scheduler spawn/restart the broker replays active workloads and telemetry before discovery, then resumes all three live feeds. +**`nvpair-job-scheduler`** consumes the accepted workload stream, compact GPU +telemetry, and discovery snapshot. It smooths fresh utilization into pressure +0–3, uses neutral pressure 1 for invalid/missing/older-than-10-second samples, +and orders by `pending + gpuPressure`, then pressure, then stable UUID. Load is +node-wide across Ollama, LM Studio, and llama.cpp because they normally contend +for the same resources. Each engine-specific `schedule:priority` carries +`{engine,nodes,ranks}` and refreshes when order, pending counts, or pressure +changes. The broker caches, generation-orders, and replays the full +`{nodes,ranks}` snapshot to the matching facade, where a newly delivered +snapshot resets optimistic reservation deltas. On scheduler spawn/restart the +broker replays active workloads and telemetry before discovery, then resumes all +three live feeds. `schedule:priority` and `node/set-priority` are internal worker contracts: the broker does not expose either notification to its connected client. @@ -205,7 +241,11 @@ Two classes of proxy notification are **not** re-emitted under the `ollama-proxy **Opt-in.** Only delivered to a peer that has called `workloads:subscribe`; silent otherwise. Once subscribed, the broker pushes a `workloads:upsert` whenever a workload is created or its state changes, and a `workloads:remove` when one is retired. The stream is the union of two sources, in the same shape regardless of origin: -- **Local workloads** — the `workload:*` lifecycle events the supervised Ollama and LM Studio proxies emit per inference request, stamped with this host's stable `hostUuid` (`originatedFrom`) and translated to `workloads:upsert`. The proxy also fills in `scheduledOn` with the destination node's `hostUuid`; the broker passes that through unchanged. +- **Local workloads** — the `workload:*` lifecycle events the enabled proxy + facades emit per inference request, stamped with this host's stable `hostUuid` + (`originatedFrom`) and translated to `workloads:upsert`. The proxy also fills + in `scheduledOn` with the destination node's `hostUuid`; the broker passes + that through unchanged. - **Peer workloads** — the `workloads:upsert` / `workloads:remove` the `nvpair-workload-manager` relays from other nodes after validating and de-duplicating their broadcasts. - **Inferred workloads** — a `workloads:upsert` transitioning a workload to `failed` that **no origin ever sent**. The broker synthesizes one in two situations: when a node leaves discovery while workloads are pinned to it, and when a remote origin that is still present stops re-asserting a workload this node believes is running (the origin's re-sync heartbeat asserts each of its active workloads indefinitely, so prolonged silence about one means it is finished or the origin is gone). Both are recorded as *inferred*, so the origin's next authoritative event overrides them; a client should treat a `failed` as the broker's best current answer rather than proof the origin reported a failure, and its `error` text names the reason. Workloads this node originated or is itself executing are never inferred about. @@ -569,7 +609,11 @@ Attach to a pre-existing endpoint: ## What this version intentionally does NOT do (yet) -- **Engine-advertise control surface.** Engine registration is auto-driven only: the broker tracks local ollama / LM Studio on their fixed coordinates and registers `ol` / `lm` with the daemon while up. There's no manual-advertise RPC (custom service, port, name, or TXT), and no way to advertise anything other than the detected engines. +- **Engine-advertise control surface.** Engine registration is auto-driven only: + the broker tracks local Ollama, LM Studio, and llama.cpp and registers `ol`, + `lm`, or `lc` with the daemon while each is up. There's no manual-advertise + RPC (custom service, port, name, or TXT), and no way to advertise anything + other than the detected engines. - **node-info control surface.** node-info is spawned and torn down with the broker, and the broker pushes it only two things over stdin: the log level, and this node's cluster principal (`nodeinfo:set-cluster-identity`, sent on spawn and on every membership or pin-set change, because node-info holds no cluster dir and so cannot read membership itself). Otherwise it's hands-off: the broker registers its port with the daemon (which enriches over plain HTTP) but doesn't pass through TLS material (`--cert` / `--key` / `--client-ca`) or a custom `--port`, and exposes no RPC to query or reconfigure it. It runs with its own defaults plus those two pushes. - **Manual-node persistence across restarts.** `nvpair-manual-nodes` keeps its entries only in memory and the broker holds no authoritative copy, so a manual-nodes crash-and-restart loses the user's manual nodes (the broker evicts the orphaned entries from the snapshot; clients must re-add them). - **Per-event push semantics.** `discovery:nodes-changed` always carries the full current snapshot, not a delta. For small N this is fine and lets the client treat the payload as authoritative without state reconciliation. `errors:update` is likewise a full snapshot. diff --git a/services/nvpair-ui-broker/engineproxy.go b/services/nvpair-ui-broker/engineproxy.go index 0163bd67..036370ca 100644 --- a/services/nvpair-ui-broker/engineproxy.go +++ b/services/nvpair-ui-broker/engineproxy.go @@ -154,9 +154,9 @@ func (b *Broker) engineProxy(p engineProxyProfile) *engineProxyRuntime { return b.engineProxies[p.Name] } -// ollamaState and lmstudioState are shorthand for the two engines this build -// ships, for code that is inherently about one of them. Profile-generic code -// should take an engineProxyProfile and call engineProxy instead. +// ollamaState and lmstudioState are shorthands for code that is inherently +// about those engines' special ownership behavior. Profile-generic code should +// take an engineProxyProfile and call engineProxy instead. func (b *Broker) ollamaState() *engineProxyRuntime { return b.engineProxy(ollamaProxyProfile) } func (b *Broker) lmstudioState() *engineProxyRuntime { return b.engineProxy(lmstudioProxyProfile) } diff --git a/services/readme.md b/services/readme.md index 4ce9b5d9..9e1d9634 100644 --- a/services/readme.md +++ b/services/readme.md @@ -11,9 +11,10 @@ local network: each node advertises itself over mDNS as one consolidated node offers and where to reach them. What a discovered node can actually serve is a separate question, answered after -discovery. A node may be running [Ollama](https://ollama.com/), LM Studio, both, -or neither, and its model inventory is fetched over HTTP from its engine-manager -rather than crammed into mDNS TXT records, which are too small to carry it. +discovery. A node may be running [Ollama](https://ollama.com/), LM Studio, +llama.cpp, any combination, or none, and its model inventory is fetched over +HTTP from its engine-manager rather than crammed into mDNS TXT records, which +are too small to carry it. Locally, each node exposes compatibility proxies — Ollama-compatible and OpenAI-compatible — so an unmodified client on that machine can reach any capable @@ -27,6 +28,12 @@ can serve that loopback-only address from the same router when it is free — `localhost` is claimed on IPv4 and IPv6 together, and remote or HTTPS targets are never intercepted. +The backend's llama.cpp facade is currently opt-in: launch the broker with +`--proxy-engines ollama,lmstudio,llamacpp`. It exposes OpenAI-compatible traffic +on `http://localhost:8080` while the managed router runs on `8081`. This +checkpoint does not yet add llama.cpp proxy/model controls to the TUI or any +desktop workflow. + PAIR ships a graphical UI alongside these services. The UI launches **`nvpair-ui-broker`** from the same directory; the broker orchestrates the workers and exposes a newline-delimited JSON-RPC 2.0 API over stdio (or a Unix @@ -62,10 +69,11 @@ configuration. The broker feeds every accepted local or peer workload transition plus compact GPU telemetry to the scheduler. Queued and running work is counted by destination -node across Ollama and LM Studio together. Fresh maximum-GPU utilization is -smoothed into pressure 0–3; missing or stale telemetry is neutral. Rankings use -`pending + gpuPressure`, and each proxy adds local reservations before choosing, -so bursts spread without waiting for workload feedback. +node across Ollama, LM Studio, and llama.cpp together. Fresh maximum-GPU +utilization is smoothed into pressure 0–3; missing or stale telemetry is +neutral. Rankings use `pending + gpuPressure`, and each facade adds local +reservations before choosing, so bursts spread without waiting for workload +feedback. ## Repository layout From f34317d3e2260da26dfb4603efdef1658d635a72 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 13:59:25 -0700 Subject: [PATCH 12/70] feat(tui): select opt-in proxy engines Signed-off-by: Sherief Farouk --- docs/terminal-interface.mdx | 10 ++-- services/nvpair-tui/README.md | 7 ++- services/nvpair-tui/main.go | 11 ++++- services/nvpair-tui/proxyengines.go | 49 +++++++++++++++++++ services/nvpair-tui/proxyengines_test.go | 45 +++++++++++++++++ services/nvpair-tui/supervisor.go | 9 +++- services/nvpair-tui/supervisor_test.go | 15 +++++- services/nvpair-tui/ui/engines.go | 7 ++- services/nvpair-tui/ui/proxies.go | 22 ++++----- services/nvpair-tui/ui/proxies_test.go | 61 ++++++++++++++++++++---- services/nvpair-tui/ui/ui.go | 17 ++++--- services/readme.md | 10 ++-- 12 files changed, 218 insertions(+), 45 deletions(-) create mode 100644 services/nvpair-tui/proxyengines.go create mode 100644 services/nvpair-tui/proxyengines_test.go diff --git a/docs/terminal-interface.mdx b/docs/terminal-interface.mdx index 1eb99b5f..ae94285e 100644 --- a/docs/terminal-interface.mdx +++ b/docs/terminal-interface.mdx @@ -64,9 +64,13 @@ cd | Flag | Effect | | --- | --- | | `--broker-path ` | Use a service binary that is not beside `nvpair-tui` | +| `--proxy-engines ` | Select the proxy facades to start and show. Defaults to `ollama,lmstudio`; add `llamacpp` to opt in to llama.cpp | | `--log-level ` | Verbosity of the terminal interface's own logging: `debug`, `info`, `warn`, or `error`. PAIR also reads this from `NVPAIR_LOG_LEVEL` | | `--version` | Print the version and exit | +For example, start and display all three current facades with +`nvpair-tui --proxy-engines ollama,lmstudio,llamacpp`. + The interface's own log output goes to stderr, so it never corrupts the display. Service logs appear on the **Logs** tab instead. @@ -130,7 +134,7 @@ Tab switching and `q` do not work until you do. | 1 | **Overview** | Service uptime and version, and an `ok` / `DOWN` table for each worker | | 2 | **Errors** | Active service errors by severity, age, node, and message | | 3 | **Nodes** | Nodes discovered on the network, with `Connected` or `In cluster` status | -| 4 | **Proxies** | Both compatible proxies: listening port, discovered upstreams, and which node is selected | +| 4 | **Proxies** | Selected compatible facades: listening port, discovered upstreams, and which node is selected | | 5 | **Workloads** | Live inference workloads: ID, model, engine, state, and age | | 6 | **Engines** | Local engines: installed, running, healthy, and port | | 7 | **Cluster** | This node's identity, cluster membership, and pairing | @@ -214,8 +218,8 @@ engine, and state. The **Proxies** tab (4) shows each proxy's listening port and whether it is routing automatically (`selected=auto`) or pinned to one node. Press `g` to -switch between the two engines, `enter` to pin the highlighted upstream, and `a` -to return to automatic routing. Leave it on automatic unless you are +switch between the selected engines, `enter` to pin the highlighted upstream, +and `a` to return to automatic routing. Leave it on automatic unless you are deliberately testing one node. The **Overview** tab (1) reports whether each worker is up. Worker status is diff --git a/services/nvpair-tui/README.md b/services/nvpair-tui/README.md index 3963d1a8..539a1aea 100644 --- a/services/nvpair-tui/README.md +++ b/services/nvpair-tui/README.md @@ -30,7 +30,7 @@ Tabs: | **Overview** | Broker liveness/version/uptime (`ping`) and a per-worker health table derived from the broker's `supervisor:subprocess-crashed:*` errors. | | **Errors** | The service-error datastore (`errors:get-initial` + live `errors:update`); `c` clears the selected entry. | | **Nodes** | mDNS-discovered Ollama nodes (`discovery:subscribe` / `discovery:nodes-changed`). | -| **Proxies** | Ollama and LM Studio reverse proxies: status, discovered upstreams, select a node (`enter`/`a`), set the listen port (`p`). | +| **Proxies** | Selected reverse-proxy facades: status, discovered upstreams, select a node (`enter`/`a`), set the listen port (`p`). Defaults to Ollama and LM Studio. | | **Workloads** | Live cluster workloads (`workloads:subscribe` / `workloads:upsert` / `workloads:remove`). | | **Engines** | Local inference engines: install (`i`), start (`s`), stop (`x`), restart (`r`), uninstall (`u`). | | **Cluster** | Pairing + membership: invite by address (`i`, shows the six-digit PIN — the first invite auto-founds a cluster of one), accept (`a`) / decline (`d`) an inbound invite, remove a member (`r`), leave (`L`). | @@ -55,10 +55,15 @@ installed `bin/` layout). Override with `--broker-path`: ```sh nvpair-tui # broker is a sibling binary nvpair-tui --broker-path /opt/nvpair/bin/nvpair-ui-broker +nvpair-tui --proxy-engines ollama,lmstudio,llamacpp nvpair-tui --log-level debug # own logging (to stderr) nvpair-tui --version ``` +`--proxy-engines` accepts canonical engine ids from the shared engine table, +passes the same selection to the broker, and builds the Proxies view from it. +llama.cpp remains opt-in at this checkpoint. + Logging goes to stderr (the broker's logs are shown inside the **Logs** tab, not on the terminal), so it never corrupts the full-screen UI. diff --git a/services/nvpair-tui/main.go b/services/nvpair-tui/main.go index 4ee976f9..c754f40c 100644 --- a/services/nvpair-tui/main.go +++ b/services/nvpair-tui/main.go @@ -31,6 +31,7 @@ var Version = "dev" func main() { brokerPath := flag.String("broker-path", "", "path to nvpair-ui-broker binary (default: ./nvpair-ui-broker alongside this executable)") + proxyEnginesCSV := flag.String("proxy-engines", defaultProxyEngineCSV(), "comma-separated engines to front with a proxy") showVersion := flag.Bool("version", false, "print version and exit") resolveLevel := applog.RegisterFlag(nil, slog.LevelInfo) flag.Parse() @@ -42,6 +43,12 @@ func main() { applog.Init("nvpair-tui", resolveLevel()) + proxyEngines, err := parseProxyEngines(*proxyEnginesCSV) + if err != nil { + slog.Error("invalid --proxy-engines", "err", err) + os.Exit(2) + } + resolvedBroker, err := resolveBrokerPath(*brokerPath) if err != nil { slog.Error("cannot locate broker", "err", err) @@ -61,7 +68,7 @@ func main() { } }() - sup, err := Spawn(ctx, resolvedBroker) + sup, err := Spawn(ctx, resolvedBroker, proxyEngines) if err != nil { slog.Error("failed to start broker", "err", err) os.Exit(1) @@ -70,7 +77,7 @@ func main() { // The broker's stderr (its logs plus every worker's, prefixed) is fed // into the UI's Logs view rather than the terminal, so it never // collides with the full-screen TUI on stdout. - if err := ui.Run(sup.Client, sup.Stderr); err != nil { + if err := ui.Run(sup.Client, sup.Stderr, proxyEngines); err != nil { slog.Error("ui error", "err", err) } diff --git a/services/nvpair-tui/proxyengines.go b/services/nvpair-tui/proxyengines.go new file mode 100644 index 00000000..e9fd9932 --- /dev/null +++ b/services/nvpair-tui/proxyengines.go @@ -0,0 +1,49 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "fmt" + "strings" + + "nvpair-shared/engines" +) + +func defaultProxyEngineCSV() string { + return proxyEngineCSV(engines.ProxyDefaults()) +} + +func proxyEngineCSV(selected []engines.Engine) string { + names := make([]string, len(selected)) + for i, engine := range selected { + names[i] = engine.Name + } + return strings.Join(names, ",") +} + +// parseProxyEngines narrows user input at the TUI boundary before the selected +// engines are passed to both the broker and the Proxies view. +func parseProxyEngines(csv string) ([]engines.Engine, error) { + seen := map[string]bool{} + var selected []engines.Engine + for _, raw := range strings.Split(csv, ",") { + name := strings.TrimSpace(raw) + if name == "" { + continue + } + engine, ok := engines.ByName(name) + if !ok { + return nil, fmt.Errorf( + "unknown engine %q; known engines are %s", + name, + strings.Join(engines.Names(), ", "), + ) + } + if !seen[name] { + seen[name] = true + selected = append(selected, engine) + } + } + return selected, nil +} diff --git a/services/nvpair-tui/proxyengines_test.go b/services/nvpair-tui/proxyengines_test.go new file mode 100644 index 00000000..e36a9897 --- /dev/null +++ b/services/nvpair-tui/proxyengines_test.go @@ -0,0 +1,45 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "strings" + "testing" +) + +func TestDefaultProxyEngineCSVUsesSharedDefaults(t *testing.T) { + if got, want := defaultProxyEngineCSV(), "ollama,lmstudio"; got != want { + t.Fatalf("default proxy engines = %q, want %q", got, want) + } +} + +func TestParseProxyEnginesPreservesExplicitSelection(t *testing.T) { + got, err := parseProxyEngines("llamacpp, ollama, llamacpp") + if err != nil { + t.Fatalf("parse explicit engines: %v", err) + } + if len(got) != 2 || got[0].Name != "llamacpp" || got[1].Name != "ollama" { + t.Fatalf("parsed engines = %v, want [llamacpp ollama]", got) + } +} + +func TestParseProxyEnginesAllowsNoFacades(t *testing.T) { + got, err := parseProxyEngines(" , ") + if err != nil { + t.Fatalf("parse empty selection: %v", err) + } + if len(got) != 0 { + t.Fatalf("parsed engines = %v, want none", got) + } +} + +func TestParseProxyEnginesRejectsUnknownEngine(t *testing.T) { + _, err := parseProxyEngines("llamacpp,vllm") + if err == nil { + t.Fatal("unknown engine was accepted") + } + if !strings.Contains(err.Error(), `unknown engine "vllm"`) { + t.Fatalf("error = %q, want unknown-engine detail", err) + } +} diff --git a/services/nvpair-tui/supervisor.go b/services/nvpair-tui/supervisor.go index 546a118d..41532210 100644 --- a/services/nvpair-tui/supervisor.go +++ b/services/nvpair-tui/supervisor.go @@ -13,6 +13,7 @@ import ( "runtime" "time" + "nvpair-shared/engines" "nvpair-tui/rpc" ) @@ -77,8 +78,8 @@ func resolveBrokerPath(override string) (string, error) { // runs with its working directory set to the broker's own directory so // the broker's sibling-binary worker resolution finds nvpair-node-scanner et // al. ctx governs the client read loop; use Shutdown for an orderly stop. -func Spawn(ctx context.Context, brokerPath string) (*Supervisor, error) { - cmd := exec.Command(brokerPath) +func Spawn(ctx context.Context, brokerPath string, proxyEngines []engines.Engine) (*Supervisor, error) { + cmd := exec.Command(brokerPath, brokerArgs(proxyEngines)...) cmd.Dir = filepath.Dir(brokerPath) configureSubprocess(cmd) @@ -105,6 +106,10 @@ func Spawn(ctx context.Context, brokerPath string) (*Supervisor, error) { return &Supervisor{cmd: cmd, stdin: stdin, Client: client, Stderr: stderr}, nil } +func brokerArgs(proxyEngines []engines.Engine) []string { + return []string{"--proxy-engines", proxyEngineCSV(proxyEngines)} +} + // Shutdown asks the broker to stop cleanly: send the shutdown RPC, close // its stdin (a second, EOF-based stop signal), then wait up to // shutdownGrace before killing it. The broker tears its own workers down diff --git a/services/nvpair-tui/supervisor_test.go b/services/nvpair-tui/supervisor_test.go index ab039b31..52698fd3 100644 --- a/services/nvpair-tui/supervisor_test.go +++ b/services/nvpair-tui/supervisor_test.go @@ -12,6 +12,8 @@ import ( "path/filepath" "testing" "time" + + "nvpair-shared/engines" ) // TestMain doubles as a fake nvpair-ui-broker when NVPAIR_TUI_FAKE_BROKER=1. @@ -72,7 +74,7 @@ func TestSupervisorReadyAndShutdown(t *testing.T) { ctx, cancel := context.WithCancel(context.Background()) defer cancel() - sup, err := Spawn(ctx, os.Args[0]) + sup, err := Spawn(ctx, os.Args[0], engines.ProxyDefaults()) if err != nil { t.Fatalf("spawn: %v", err) } @@ -105,3 +107,14 @@ func TestSupervisorReadyAndShutdown(t *testing.T) { t.Fatal("shutdown did not complete") } } + +func TestBrokerArgsForwardProxySelection(t *testing.T) { + llamacpp, ok := engines.ByName("llamacpp") + if !ok { + t.Fatal("shared engine table has no llamacpp") + } + got := brokerArgs([]engines.Engine{llamacpp}) + if len(got) != 2 || got[0] != "--proxy-engines" || got[1] != "llamacpp" { + t.Fatalf("broker args = %v, want [--proxy-engines llamacpp]", got) + } +} diff --git a/services/nvpair-tui/ui/engines.go b/services/nvpair-tui/ui/engines.go index 809febfb..695e6a70 100644 --- a/services/nvpair-tui/ui/engines.go +++ b/services/nvpair-tui/ui/engines.go @@ -218,10 +218,9 @@ func (v *enginesView) handleKey(msg tea.KeyMsg) tea.Cmd { // pullParams builds the engine:action{action:"pull_model"} params for a pull. // The model name is sent under BOTH "name" and "model" — mirroring -// PullModelStream's own empty-params default — because the two engines key it -// differently: Ollama's pull_model is HTTP /api/pull (body key "name"), while -// LM Studio's is a CLI action `lms get {model}` resolved from the "model" key. -// Sending only one key silently no-ops the pull on the other engine. +// PullModelStream's own empty-params default — because bundled engines key it +// differently: Ollama's HTTP action uses "name", while LM Studio and llama.cpp +// resolve "model". Sending only one key silently no-ops a supported engine. func pullParams(engine, model string) map[string]any { return map[string]any{"engine": engine, "action": "pull_model", "params": map[string]string{"name": model, "model": model}} } diff --git a/services/nvpair-tui/ui/proxies.go b/services/nvpair-tui/ui/proxies.go index 49a3ca34..c1986cb3 100644 --- a/services/nvpair-tui/ui/proxies.go +++ b/services/nvpair-tui/ui/proxies.go @@ -25,13 +25,11 @@ type proxyNode struct { Port int `json:"port"` } -// buildProxyEngines makes one tab per default-enabled facade, in the shared -// table's order. Opt-in engines appear only when the TUI passes an explicit -// selection to the broker and this view. -func buildProxyEngines() []*proxyEngine { - defaults := engines.ProxyDefaults() - out := make([]*proxyEngine, 0, len(defaults)) - for _, e := range defaults { +// buildProxyEngines makes one tab per facade selected at the TUI boundary, in +// the same order passed to the broker. +func buildProxyEngines(selected []engines.Engine) []*proxyEngine { + out := make([]*proxyEngine, 0, len(selected)) + for _, e := range selected { out = append(out, &proxyEngine{label: e.DisplayName, prefix: e.ComponentName(), table: newTable(nil)}) } return out @@ -40,8 +38,8 @@ func buildProxyEngines() []*proxyEngine { // proxyEngine is one reverse proxy the broker fronts. They all speak the same // routing/failover contract; only the JSON-RPC prefix and label differ. type proxyEngine struct { - label string // "Ollama" / "LM Studio" - prefix string // "ollama-proxy" / "lmstudio-proxy" + label string + prefix string ready bool port int selected string @@ -49,7 +47,7 @@ type proxyEngine struct { table table.Model } -// proxiesView shows both reverse proxies: per-engine status (ready/port/ +// proxiesView shows the selected reverse proxies: per-engine status (ready/port/ // selected node) and the focused engine's discovered upstreams, with // actions to select a node and set the listen port. type proxiesView struct { @@ -94,14 +92,14 @@ var ( proxyAutoKey = key.NewBinding(key.WithKeys("a"), key.WithHelp("a", "auto-select")) ) -func newProxiesView(client *rpc.Client) *proxiesView { +func newProxiesView(client *rpc.Client, selected []engines.Engine) *proxiesView { ti := textinput.New() ti.Placeholder = "port" ti.CharLimit = 5 v := &proxiesView{ client: client, portInput: ti, - engines: buildProxyEngines(), + engines: buildProxyEngines(selected), } return v } diff --git a/services/nvpair-tui/ui/proxies_test.go b/services/nvpair-tui/ui/proxies_test.go index f0460fe9..0a23fb21 100644 --- a/services/nvpair-tui/ui/proxies_test.go +++ b/services/nvpair-tui/ui/proxies_test.go @@ -7,18 +7,61 @@ import ( "testing" "nvpair-shared/engines" + "nvpair-tui/rpc" ) -func TestBuildProxyEnginesUsesSharedDefaults(t *testing.T) { - got := buildProxyEngines() - want := engines.ProxyDefaults() - if len(got) != len(want) { - t.Fatalf("proxy tabs = %d, want %d", len(got), len(want)) +func TestBuildProxyEnginesUsesSelectedEngines(t *testing.T) { + test := func(name string, selected []engines.Engine) { + t.Run(name, func(t *testing.T) { + got := buildProxyEngines(selected) + if len(got) != len(selected) { + t.Fatalf("proxy tabs = %d, want %d", len(got), len(selected)) + } + for i, engine := range selected { + if got[i].label != engine.DisplayName || got[i].prefix != engine.ComponentName() { + t.Errorf("proxy tab %d = (%q, %q), want (%q, %q)", + i, got[i].label, got[i].prefix, engine.DisplayName, engine.ComponentName()) + } + } + }) } - for i, engine := range want { - if got[i].label != engine.DisplayName || got[i].prefix != engine.ComponentName() { - t.Errorf("proxy tab %d = (%q, %q), want (%q, %q)", - i, got[i].label, got[i].prefix, engine.DisplayName, engine.ComponentName()) + + test("shared defaults", engines.ProxyDefaults()) + llamacpp, ok := engines.ByName("llamacpp") + if !ok { + t.Fatal("shared engine table has no llamacpp") + } + test("explicit llama.cpp", []engines.Engine{llamacpp}) +} + +func TestLlamaCPPNotificationsUseSelectedFacade(t *testing.T) { + llamacpp, ok := engines.ByName("llamacpp") + if !ok { + t.Fatal("shared engine table has no llamacpp") + } + view := newProxiesView(nil, []engines.Engine{llamacpp}) + + cmd := view.handleNotification(&rpc.Message{ + Method: "llamacpp-proxy:ready", + Params: []byte(`{"port":8080}`), + }) + if cmd != nil { + t.Fatal("ready notification unexpectedly returned a command") + } + if !view.engines[0].ready || view.engines[0].port != 8080 { + t.Fatalf("llama.cpp status = ready:%v port:%d, want ready on 8080", + view.engines[0].ready, view.engines[0].port) + } + + if cmd := view.handleNotification(&rpc.Message{Method: "llamacpp-proxy:node/discovered"}); cmd == nil { + t.Fatal("llama.cpp node notification did not schedule a nodes refresh") + } +} + +func TestDefaultViewsOmitProxiesWhenNoneSelected(t *testing.T) { + for _, view := range defaultViews(nil, nil) { + if view.Title() == "Proxies" { + t.Fatal("Proxies view present with no selected proxy engines") } } } diff --git a/services/nvpair-tui/ui/ui.go b/services/nvpair-tui/ui/ui.go index 65015a7b..d87cbf65 100644 --- a/services/nvpair-tui/ui/ui.go +++ b/services/nvpair-tui/ui/ui.go @@ -7,6 +7,7 @@ import ( "bufio" "io" + "nvpair-shared/engines" "nvpair-tui/rpc" tea "github.com/charmbracelet/bubbletea" @@ -15,12 +16,12 @@ import ( // Run builds the tabbed program over a connected broker client and the // broker's stderr stream, and blocks until the user quits. The caller is // responsible for shutting the broker down afterwards. -func Run(client *rpc.Client, stderr io.Reader) error { +func Run(client *rpc.Client, stderr io.Reader, proxyEngines []engines.Engine) error { logCh := make(chan string, 2000) go scanLines(stderr, logCh) p := tea.NewProgram( - New(client, logCh, defaultViews(client)), + New(client, logCh, defaultViews(client, proxyEngines)), tea.WithAltScreen(), ) _, err := p.Run() @@ -40,17 +41,21 @@ func scanLines(r io.Reader, out chan<- string) { } // defaultViews lists the tabs in display order. -func defaultViews(client *rpc.Client) []View { - return []View{ +func defaultViews(client *rpc.Client, proxyEngines []engines.Engine) []View { + views := []View{ newHealthView(client), newErrorsView(client), newNodesView(client), - newProxiesView(client), + } + if len(proxyEngines) > 0 { + views = append(views, newProxiesView(client, proxyEngines)) + } + return append(views, newWorkloadsView(client), newEnginesView(client), newClusterView(client), newManualView(client), newSettingsView(client), newLogsView(client), - } + ) } diff --git a/services/readme.md b/services/readme.md index 9e1d9634..45440cc3 100644 --- a/services/readme.md +++ b/services/readme.md @@ -28,11 +28,11 @@ can serve that loopback-only address from the same router when it is free — `localhost` is claimed on IPv4 and IPv6 together, and remote or HTTPS targets are never intercepted. -The backend's llama.cpp facade is currently opt-in: launch the broker with -`--proxy-engines ollama,lmstudio,llamacpp`. It exposes OpenAI-compatible traffic -on `http://localhost:8080` while the managed router runs on `8081`. This -checkpoint does not yet add llama.cpp proxy/model controls to the TUI or any -desktop workflow. +The backend's llama.cpp facade is currently opt-in: launch the broker or TUI +with `--proxy-engines ollama,lmstudio,llamacpp`. It exposes OpenAI-compatible +traffic on `http://localhost:8080` while the managed router runs on `8081`. The +TUI builds its proxy controls from that selection but does not yet expose model +inventory/load/unload controls; the desktop still has no llama.cpp workflow. PAIR ships a graphical UI alongside these services. The UI launches **`nvpair-ui-broker`** from the same directory; the broker orchestrates the From c1fd011d9861c1002473cbde2428133e03a456a9 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 15:44:29 -0700 Subject: [PATCH 13/70] feat(tui): manage local engine models Signed-off-by: Sherief Farouk --- docs/terminal-interface.mdx | 22 ++- services/nvpair-tui/README.md | 1 + services/nvpair-tui/ui/models.go | 228 ++++++++++++++++++++++++++ services/nvpair-tui/ui/models_test.go | 93 +++++++++++ services/nvpair-tui/ui/ui.go | 1 + services/readme.md | 4 +- 6 files changed, 338 insertions(+), 11 deletions(-) create mode 100644 services/nvpair-tui/ui/models.go create mode 100644 services/nvpair-tui/ui/models_test.go diff --git a/docs/terminal-interface.mdx b/docs/terminal-interface.mdx index ae94285e..74db924c 100644 --- a/docs/terminal-interface.mdx +++ b/docs/terminal-interface.mdx @@ -137,10 +137,11 @@ Tab switching and `q` do not work until you do. | 4 | **Proxies** | Selected compatible facades: listening port, discovered upstreams, and which node is selected | | 5 | **Workloads** | Live inference workloads: ID, model, engine, state, and age | | 6 | **Engines** | Local engines: installed, running, healthy, and port | -| 7 | **Cluster** | This node's identity, cluster membership, and pairing | -| 8 | **Manual** | Nodes you added by address, with reachability | -| 9 | **Settings** | Node settings: force ports, cluster auto-sync, and cluster ID and name | -| 10 | **Logs** | Service log output, with live log-level control | +| 7 | **Models** | Local models by engine, including loaded/idle state | +| 8 | **Cluster** | This node's identity, cluster membership, and pairing | +| 9 | **Manual** | Nodes you added by address, with reachability | +| 10 | **Settings** | Node settings: force ports, cluster auto-sync, and cluster ID and name | +| 11 | **Logs** | Service log output, with live log-level control | ## Pair This Machine with Another @@ -155,7 +156,7 @@ Pairing is the same six-digit PIN exchange the desktop application uses. This is the easier path, because there is no address to type. Prefer it whenever PAIR has already discovered the machine you want. -**To invite a machine by address**, from the **Cluster** tab (7): +**To invite a machine by address**, from the **Cluster** tab (8): 1. Press `i`. 2. Type the other machine's host, or `host:port` if it is not on the default @@ -205,6 +206,10 @@ From the **Engines** tab (6), select an engine with `j` / `k`, then: Pressing `p` opens a prompt. Type the model name, for example `qwen4:12b`, and press `enter`. Progress appears on the status line. +The **Models** tab (7) lists every running engine's local inventory. Select a +model and press `enter` to load it into memory or `u` to unload it. State changes +come from the engine manager and appear as `loaded`, `idle`, or `unknown`. + A node can serve a request only when it is online, a compatible engine is running, and the requested model is present on that node. To route across several machines, download the same model on each. @@ -229,13 +234,13 @@ best-effort. `DOWN` means a worker reported a crash. On **Errors** (2), press `c` to clear the selected entry. -On **Logs** (10), scroll with `j` / `k` and the page keys. Set the log level for +On **Logs** (11), scroll with `j` / `k` and the page keys. Set the log level for the whole service fleet with `d` (debug), `i` (info), `w` (warn), or `e` (error). This is the first place to look when something has not started. ## Change Settings -On **Settings** (9), move with `j` / `k` and press `enter`. Booleans toggle +On **Settings** (10), move with `j` / `k` and press `enter`. Booleans toggle immediately. Text fields open for editing, with `enter` to save and `esc` to cancel. @@ -247,8 +252,7 @@ for how PAIR arranges ports. It is an operations tool, not a full replacement for the desktop application: -- It cannot list or delete models. You can download one, but the interface shows - no model inventory. +- It cannot delete models. - It cannot change an engine's port. The port column is read-only. Use the desktop application to change it. - It cannot update an engine or control engines on other cluster nodes. diff --git a/services/nvpair-tui/README.md b/services/nvpair-tui/README.md index 539a1aea..46cb67db 100644 --- a/services/nvpair-tui/README.md +++ b/services/nvpair-tui/README.md @@ -33,6 +33,7 @@ Tabs: | **Proxies** | Selected reverse-proxy facades: status, discovered upstreams, select a node (`enter`/`a`), set the listen port (`p`). Defaults to Ollama and LM Studio. | | **Workloads** | Live cluster workloads (`workloads:subscribe` / `workloads:upsert` / `workloads:remove`). | | **Engines** | Local inference engines: install (`i`), start (`s`), stop (`x`), restart (`r`), uninstall (`u`). | +| **Models** | Local model inventory and loaded/idle/unknown state for every running engine; load (`enter`) or unload (`u`) the selected model. | | **Cluster** | Pairing + membership: invite by address (`i`, shows the six-digit PIN — the first invite auto-founds a cluster of one), accept (`a`) / decline (`d`) an inbound invite, remove a member (`r`), leave (`L`). | | **Manual** | User-added nodes: add by address (`a`), remove (`r`). | | **Settings** | The node-settings store (force-ports, cluster auto-sync, cluster id/name). | diff --git a/services/nvpair-tui/ui/models.go b/services/nvpair-tui/ui/models.go new file mode 100644 index 00000000..0a75b6ff --- /dev/null +++ b/services/nvpair-tui/ui/models.go @@ -0,0 +1,228 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package ui + +import ( + "slices" + + "nvpair-tui/rpc" + + "github.com/charmbracelet/bubbles/key" + "github.com/charmbracelet/bubbles/table" + tea "github.com/charmbracelet/bubbletea" +) + +type modelInventory struct { + ByEngine map[string][]string `json:"modelsByEngine"` + LoadedByEngine map[string][]string `json:"loadedByEngine"` +} +type modelRow struct { + engine, model, state string +} +type modelsView struct { + client *rpc.Client + table table.Model + rows []modelRow + status string +} +type modelsLoadedMsg struct { + inventory modelInventory + err error +} +type modelActionMsg struct { + what string + row modelRow + err error +} +type modelActionRequest struct { + Engine string `json:"engine"` + Action string `json:"action"` + Params modelActionParams `json:"params"` +} +type modelActionParams struct { + Model string `json:"model"` + Stream *bool `json:"stream,omitempty"` + KeepAlive *int `json:"keep_alive,omitempty"` +} + +var ( + modelLoadKey = key.NewBinding(key.WithKeys("enter"), key.WithHelp("enter", "load")) + modelUnloadKey = key.NewBinding(key.WithKeys("u"), key.WithHelp("u", "unload")) +) + +func newModelsView(client *rpc.Client) *modelsView { + view := &modelsView{client: client, table: newTable(nil)} + view.SetSize(80, 20) + return view +} +func (v *modelsView) Title() string { return "Models" } + +func (v *modelsView) Init() tea.Cmd { + return tea.Batch( + call(v.client, "engine:subscribe", nil, func(_ *rpc.Message, _ error) tea.Msg { return nil }), + v.loadCmd(), + ) +} + +func (v *modelsView) loadCmd() tea.Cmd { + return call(v.client, "engine:models", nil, func(msg *rpc.Message, err error) tea.Msg { + if err != nil { + return modelsLoadedMsg{err: err} + } + var inventory modelInventory + err = decodeParams(msg.Result, &inventory) + return modelsLoadedMsg{inventory: inventory, err: err} + }) +} + +func (v *modelsView) SetSize(width, height int) { + const engineWidth, stateWidth = 14, 8 + v.table.SetColumns([]table.Column{ + {Title: "ENGINE", Width: engineWidth}, + {Title: "MODEL", Width: clampWidth(width-engineWidth-stateWidth-2, 16)}, + {Title: "STATE", Width: stateWidth}, + }) + v.table.SetWidth(width) + v.table.SetHeight(clampWidth(height-2, 1)) +} + +func (v *modelsView) Update(msg tea.Msg) tea.Cmd { + switch msg := msg.(type) { + case modelsLoadedMsg: + if msg.err != nil { + v.status = "load models failed: " + msg.err.Error() + } else { + v.status = "" + v.apply(msg.inventory) + } + return nil + case modelActionMsg: + if msg.err != nil { + v.status = msg.what + " " + msg.row.engine + "/" + msg.row.model + " failed: " + msg.err.Error() + } else { + v.status = msg.what + " " + msg.row.engine + "/" + msg.row.model + " ok" + } + return nil + case NotificationMsg: + if msg.Msg.Method == "engine:state-changed" { + return v.loadCmd() + } + if msg.Msg.Method != "engine:models-changed" { + return nil + } + var changed struct { + Models modelInventory `json:"models"` + } + if err := decodeParams(msg.Msg.Params, &changed); err != nil { + v.status = "update models failed: " + err.Error() + } else { + v.status = "" + v.apply(changed.Models) + } + return nil + case tea.KeyMsg: + return v.handleKey(msg) + } + return nil +} + +func (v *modelsView) handleKey(msg tea.KeyMsg) tea.Cmd { + what := "" + switch { + case key.Matches(msg, modelLoadKey): + what = "load" + case key.Matches(msg, modelUnloadKey): + what = "unload" + default: + var cmd tea.Cmd + v.table, cmd = v.table.Update(msg) + return cmd + } + row, ok := v.selectedModel() + if !ok { + return nil + } + v.status = what + " " + row.engine + "/" + row.model + "..." + request := newModelActionRequest(row, what) + return call(v.client, "engine:action", request, func(_ *rpc.Message, err error) tea.Msg { + return modelActionMsg{what: what, row: row, err: err} + }) +} + +func newModelActionRequest(row modelRow, what string) modelActionRequest { + request := modelActionRequest{ + Engine: row.engine, + Action: what + "_model", + Params: modelActionParams{Model: row.model}, + } + if row.engine != "ollama" { + return request + } + switch what { + case "load": + stream := false + request.Action = "run_model" + request.Params.Stream = &stream + case "unload": + keepAlive := 0 + request.Params.KeepAlive = &keepAlive + } + return request +} + +func (v *modelsView) selectedModel() (modelRow, bool) { + index := v.table.Cursor() + if index < 0 || index >= len(v.rows) { + return modelRow{}, false + } + return v.rows[index], true +} + +func (v *modelsView) apply(inventory modelInventory) { + engines := make([]string, 0, len(inventory.ByEngine)) + for engine := range inventory.ByEngine { + engines = append(engines, engine) + } + slices.Sort(engines) + + var rows []modelRow + var tableRows []table.Row + for _, engine := range engines { + models := append([]string(nil), inventory.ByEngine[engine]...) + slices.Sort(models) + for _, model := range models { + loadedModels, known := inventory.LoadedByEngine[engine] + state := "unknown" + if known { + state = "idle" + if slices.Contains(loadedModels, model) { + state = "loaded" + } + } + row := modelRow{engine: engine, model: model, state: state} + rows = append(rows, row) + tableRows = append(tableRows, table.Row{engine, model, state}) + } + } + v.rows = rows + v.table.SetRows(tableRows) +} + +func (v *modelsView) View() string { + if len(v.rows) == 0 { + if v.status != "" { + return statusErrStyle.Render(v.status) + } + return footerStyle.Render("No local models reported by running engines.") + } + out := v.table.View() + if v.status != "" { + out += "\n" + footerStyle.Render(v.status) + } + return out +} + +func (v *modelsView) Help() []key.Binding { + return []key.Binding{modelLoadKey, modelUnloadKey} +} diff --git a/services/nvpair-tui/ui/models_test.go b/services/nvpair-tui/ui/models_test.go new file mode 100644 index 00000000..a0956546 --- /dev/null +++ b/services/nvpair-tui/ui/models_test.go @@ -0,0 +1,93 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package ui + +import ( + "errors" + "strings" + "testing" + + "nvpair-tui/rpc" +) + +func TestModelsViewLoadsEveryEngineAndResidency(t *testing.T) { + var inventory modelInventory + raw := []byte(`{"modelsByEngine":{"ollama":["b:latest","a:latest"],"llamacpp":["owner/model-GGUF:Q4_K_M"]},"loadedByEngine":{"ollama":["b:latest"],"llamacpp":["owner/model-GGUF:Q4_K_M"]}}`) + if err := decodeParams(raw, &inventory); err != nil { + t.Fatalf("decode baseline: %v", err) + } + view := newModelsView(nil) + view.Update(modelsLoadedMsg{inventory: inventory}) + if len(view.rows) != 3 { + t.Fatalf("model rows = %v, want three rows", view.rows) + } + if row := view.rows[0]; row.engine != "llamacpp" || row.model != "owner/model-GGUF:Q4_K_M" || row.state != "loaded" { + t.Errorf("first row = %+v, want loaded llama.cpp model", row) + } + if row := view.rows[1]; row.engine != "ollama" || row.model != "a:latest" || row.state != "idle" { + t.Errorf("second row = %+v, want idle Ollama a:latest", row) + } + if row := view.rows[2]; row.model != "b:latest" || row.state != "loaded" { + t.Errorf("third row = %+v, want loaded Ollama b:latest", row) + } + view.table.SetCursor(1) + selected, ok := view.selectedModel() + if !ok || selected != view.rows[1] { + t.Fatalf("selected model = %+v, %v; want second row", selected, ok) + } +} + +func TestModelsChangedReplacesResidencySnapshot(t *testing.T) { + view := newModelsView(nil) + view.apply(modelInventory{ + ByEngine: map[string][]string{"llamacpp": {"owner/model:Q4_K_M"}}, + LoadedByEngine: map[string][]string{"llamacpp": {}}, + }) + view.Update(NotificationMsg{Msg: &rpc.Message{ + Method: "engine:models-changed", + Params: []byte(`{"engine":"llamacpp","models":{"modelsByEngine":{"llamacpp":["owner/model:Q4_K_M"]},"loadedByEngine":{"llamacpp":["owner/model:Q4_K_M"]}}}`), + }}) + if len(view.rows) != 1 || view.rows[0].state != "loaded" { + t.Fatalf("rows after residency push = %+v, want one loaded model", view.rows) + } +} + +func TestModelActionRequestsMatchEngineContracts(t *testing.T) { + test := func(name, engine, what, wantAction string, wantStream, wantKeepAlive bool) { + t.Run(name, func(t *testing.T) { + request := newModelActionRequest(modelRow{engine: engine, model: "owner/model"}, what) + if request.Engine != engine || request.Action != wantAction || request.Params.Model != "owner/model" { + t.Fatalf("request = %+v, want %s %s for owner/model", request, engine, wantAction) + } + if (request.Params.Stream != nil) != wantStream { + t.Errorf("stream presence = %v, want %v", request.Params.Stream != nil, wantStream) + } + if request.Params.Stream != nil && *request.Params.Stream { + t.Error("Ollama load must send stream=false") + } + if (request.Params.KeepAlive != nil) != wantKeepAlive { + t.Errorf("keep_alive presence = %v, want %v", request.Params.KeepAlive != nil, wantKeepAlive) + } + if request.Params.KeepAlive != nil && *request.Params.KeepAlive != 0 { + t.Errorf("keep_alive = %d, want 0", *request.Params.KeepAlive) + } + }) + } + test("llama.cpp load", "llamacpp", "load", "load_model", false, false) + test("llama.cpp unload", "llamacpp", "unload", "unload_model", false, false) + test("Ollama load", "ollama", "load", "run_model", true, false) + test("Ollama unload", "ollama", "unload", "unload_model", false, true) +} + +func TestModelsViewEmptyAndErrorStates(t *testing.T) { + view := newModelsView(nil) + if got := view.View(); !strings.Contains(got, "No local models") { + t.Fatalf("empty view = %q, want empty-state guidance", got) + } + + view.Update(modelsLoadedMsg{err: errors.New("manager unavailable")}) + if got := view.View(); !strings.Contains(got, "manager unavailable") { + t.Fatalf("error view = %q, want backend error", got) + } +} diff --git a/services/nvpair-tui/ui/ui.go b/services/nvpair-tui/ui/ui.go index d87cbf65..02c2b9af 100644 --- a/services/nvpair-tui/ui/ui.go +++ b/services/nvpair-tui/ui/ui.go @@ -53,6 +53,7 @@ func defaultViews(client *rpc.Client, proxyEngines []engines.Engine) []View { return append(views, newWorkloadsView(client), newEnginesView(client), + newModelsView(client), newClusterView(client), newManualView(client), newSettingsView(client), diff --git a/services/readme.md b/services/readme.md index 45440cc3..81883129 100644 --- a/services/readme.md +++ b/services/readme.md @@ -31,8 +31,8 @@ never intercepted. The backend's llama.cpp facade is currently opt-in: launch the broker or TUI with `--proxy-engines ollama,lmstudio,llamacpp`. It exposes OpenAI-compatible traffic on `http://localhost:8080` while the managed router runs on `8081`. The -TUI builds its proxy controls from that selection but does not yet expose model -inventory/load/unload controls; the desktop still has no llama.cpp workflow. +TUI builds proxy controls from that selection and shows local inventory with +load/unload actions; the desktop still has no llama.cpp workflow. PAIR ships a graphical UI alongside these services. The UI launches **`nvpair-ui-broker`** from the same directory; the broker orchestrates the From 1668c688932174aa1ed1c1eac9b7c23ce22e5dc6 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 15:59:12 -0700 Subject: [PATCH 14/70] feat(desktop): declare the llama.cpp engine Signed-off-by: Sherief Farouk --- desktop/src/shared/constants/engines.ts | 26 +++++---- .../src/ui/constants/engine-capabilities.ts | 17 ++++++ desktop/src/ui/constants/welcome.ts | 3 +- desktop/tests/modular/engine-identity.test.ts | 57 +++++++++++++++++++ 4 files changed, 90 insertions(+), 13 deletions(-) create mode 100644 desktop/tests/modular/engine-identity.test.ts diff --git a/desktop/src/shared/constants/engines.ts b/desktop/src/shared/constants/engines.ts index 5904ae61..8cf4cad7 100644 --- a/desktop/src/shared/constants/engines.ts +++ b/desktop/src/shared/constants/engines.ts @@ -4,39 +4,41 @@ import { EngineType, ModelExpiry } from '@/shared/types/engines' // The engines `nvpair-engine-manager` ships a manifest for, and therefore the -// only ones PAIR can install, run or route to. llama-cpp, whisper-cpp, -// piper-tts, sherpa-onnx-tts and stable-diffusion-cpp were carried here as -// never-enabled placeholders; they were removed with the chat window, which was -// their only in-app consumer. Adding an engine back means shipping its manifest -// first -- an engine row without one renders commands that fail with `-32000`. -export const EngineTypes = ['ollama', 'lm-studio'] as const +// only ones PAIR can install, run or route to. Keep staged engines out of +// EnabledEngineTypes until every desktop consumer is ready to expose them. +export const EngineTypes = ['ollama', 'lm-studio', 'llama-cpp'] as const // Kept as a distinct export so a future engine can ship behind it rather than // appearing the moment its type exists. export const EnabledEngineTypes: EngineType[] = ['ollama', 'lm-studio'] as const /** - * How `nvpair-engine-manager` spells each engine on the wire. Only LM Studio - * differs from PAIR's `EngineType`; every other engine is identical on both - * sides. This is the single translation table — use `engineManagerName()` and + * How `nvpair-engine-manager` spells each engine on the wire. This is the + * single translation table — use `engineManagerName()` and * `engineTypeFromManagerName()` rather than re-deriving it from a literal. */ export const EngineManagerNames = { ollama: 'ollama', - 'lm-studio': 'lmstudio' + 'lm-studio': 'lmstudio', + 'llama-cpp': 'llamacpp' } as const satisfies Record export const EngineSources = ['bundled', 'detected', 'installed'] as const export const EngineDisplayNames: Record = { ollama: 'Ollama', - 'lm-studio': 'LM Studio' + 'lm-studio': 'LM Studio', + 'llama-cpp': 'llama.cpp' } as const /** Default docs/install URLs for built-in backends. Single source of truth for UI and adapter buildInfo(). */ export const EngineDefaultLinks: Record = { ollama: { docsUrl: 'https://docs.ollama.com/', installUrl: 'https://ollama.com/download' }, - 'lm-studio': { docsUrl: 'https://lmstudio.ai/docs', installUrl: 'https://lmstudio.ai/' } + 'lm-studio': { docsUrl: 'https://lmstudio.ai/docs', installUrl: 'https://lmstudio.ai/' }, + 'llama-cpp': { + docsUrl: 'https://github.com/ggml-org/llama.cpp/tree/master/docs', + installUrl: 'https://github.com/ggml-org/llama.cpp/releases' + } } as const export const ModelItemStatuses = ['idle', 'loading', 'loaded', 'ejecting', 'pulling'] as const diff --git a/desktop/src/ui/constants/engine-capabilities.ts b/desktop/src/ui/constants/engine-capabilities.ts index 77e6f2bf..13b223f2 100644 --- a/desktop/src/ui/constants/engine-capabilities.ts +++ b/desktop/src/ui/constants/engine-capabilities.ts @@ -43,5 +43,22 @@ export const EngineCapabilities: Record = { // server. Deleting therefore interrupts inference and needs a warning. restartsOnModelDelete: true, engineHub: { label: 'LM Studio', url: 'https://lmstudio.ai/models' } + }, + 'llama-cpp': { + hasExpiry: false, + hasEject: true, + hasInstall: ['win32', 'darwin', 'linux'], + hasEnginePort: true, + hasInstallPath: false, + hasProxyWebUI: false, + hasPreferredNode: false, + hasCrashAlert: false, + hasModelSearchOnlyWhenRunning: true, + modelOpsWhenStopped: false, + hasDeleteModel: false, + engineHub: { + label: 'llama.cpp', + url: 'https://huggingface.co/models?library=gguf' + } } } diff --git a/desktop/src/ui/constants/welcome.ts b/desktop/src/ui/constants/welcome.ts index 4f9fa04a..73142903 100644 --- a/desktop/src/ui/constants/welcome.ts +++ b/desktop/src/ui/constants/welcome.ts @@ -13,7 +13,8 @@ export const WELCOME_STEP_SUB_HEADINGS = ['', 'You can update later by clicking export const WELCOME_ENGINE_DEFAULT_SELECTED: Record = { ollama: true, - 'lm-studio': true + 'lm-studio': true, + 'llama-cpp': false } export function getWelcomeEngineCandidates(os: PlatformDisplayName): EngineType[] { diff --git a/desktop/tests/modular/engine-identity.test.ts b/desktop/tests/modular/engine-identity.test.ts new file mode 100644 index 00000000..6fd6dec8 --- /dev/null +++ b/desktop/tests/modular/engine-identity.test.ts @@ -0,0 +1,57 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +import { describe, expect, it } from 'vitest' +import { + EnabledEngineTypes, + EngineDefaultLinks, + EngineDisplayNames, + EngineManagerNames, + EngineTypes +} from '@/shared/constants/engines' +import { engineManagerName, engineTypeFromManagerName, isEngineType } from '@/shared/utils/engines' +import { EngineCapabilities } from '@/ui/constants/engine-capabilities' +import { getWelcomeEngineCandidates, WELCOME_ENGINE_DEFAULT_SELECTED } from '@/ui/constants/welcome' + +describe('engine identity', () => { + it('declares llama.cpp without enabling its desktop workflows', () => { + expect(EngineTypes).toEqual(['ollama', 'lm-studio', 'llama-cpp']) + expect(EnabledEngineTypes).toEqual(['ollama', 'lm-studio']) + expect(getWelcomeEngineCandidates('Windows')).not.toContain('llama-cpp') + expect(getWelcomeEngineCandidates('MacOS')).not.toContain('llama-cpp') + expect(getWelcomeEngineCandidates('Linux')).not.toContain('llama-cpp') + expect(WELCOME_ENGINE_DEFAULT_SELECTED['llama-cpp']).toBe(false) + }) + + it('maps the desktop id to the sole engine-manager wire id', () => { + expect(EngineManagerNames['llama-cpp']).toBe('llamacpp') + expect(engineManagerName('llama-cpp')).toBe('llamacpp') + expect(engineTypeFromManagerName('llamacpp')).toBe('llama-cpp') + expect(isEngineType('llama-cpp')).toBe(true) + }) + + it('provides complete display metadata and capabilities', () => { + for (const engineType of EngineTypes) { + expect(EngineDisplayNames[engineType]).toBeTruthy() + expect(EngineDefaultLinks[engineType].docsUrl).toMatch(/^https:\/\//) + expect(EngineDefaultLinks[engineType].installUrl).toMatch(/^https:\/\//) + expect(EngineCapabilities[engineType]).toBeDefined() + } + + expect(EngineDisplayNames['llama-cpp']).toBe('llama.cpp') + expect(EngineCapabilities['llama-cpp']).toMatchObject({ + hasExpiry: false, + hasEject: true, + hasInstall: ['win32', 'darwin', 'linux'], + hasEnginePort: true, + hasInstallPath: false, + hasProxyWebUI: false, + hasPreferredNode: false, + hasCrashAlert: false, + hasModelSearchOnlyWhenRunning: true, + modelOpsWhenStopped: false, + hasDeleteModel: false, + engineHub: { label: 'llama.cpp' } + }) + }) +}) From 9ed0df98f2b01b3ff68e3f05359515accbe7d5e6 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 16:08:01 -0700 Subject: [PATCH 15/70] refactor(desktop): centralize proxy identity mapping Signed-off-by: Sherief Farouk --- .../electron/service-bridge/modular-state.ts | 62 +++++--------- .../service-bridge/modular-supervisor.ts | 52 +++++------- .../electron/service-bridge/proxy-engines.ts | 58 +++++++++++++ .../src/shared/constants/modular-binaries.ts | 2 +- .../modular/proxy-engine-identity.test.ts | 84 +++++++++++++++++++ 5 files changed, 183 insertions(+), 75 deletions(-) create mode 100644 desktop/src/electron/service-bridge/proxy-engines.ts create mode 100644 desktop/tests/modular/proxy-engine-identity.test.ts diff --git a/desktop/src/electron/service-bridge/modular-state.ts b/desktop/src/electron/service-bridge/modular-state.ts index 151f1745..eef2ad26 100644 --- a/desktop/src/electron/service-bridge/modular-state.ts +++ b/desktop/src/electron/service-bridge/modular-state.ts @@ -35,33 +35,20 @@ import { emitBridgePush } from './broadcaster' import { mergePullProgressPercent } from './pull-error-handling' import type { JsonObject, JsonRpcNotification, JsonValue } from './json-rpc-subprocess' import { serviceLogLevel } from './service-log-level' -// Live node sources are the two reverse proxies, relayed through the broker, -// and the broker's consolidated discovery snapshot. Electron does not consume +import { + isProxyEngine, + PROXY_ENGINES, + proxyEngineFromSource, + proxySourceForEngine, + type ProxyEngine, + type ProxyNodeSource +} from './proxy-engines' + +// Live node sources are the reverse proxies, relayed through the broker, and +// the broker's consolidated discovery snapshot. Electron does not consume // worker discovery protocols directly. -type ProxyNodeSource = 'ollama-proxy' | 'lmstudio-proxy' type BrokerNodeSource = ProxyNodeSource | 'broker' -/** - * The engine proxies, by the one name that identifies each of them everywhere: - * the broker's relay prefix, the errors-pipeline source, and the node source - * recorded here. That is `ComponentName` in `services/shared/engines`, always - * `-proxy`. - */ -export const PROXY_NODE_SOURCES: readonly ProxyNodeSource[] = ['ollama-proxy', 'lmstudio-proxy'] - -/** - * Engines surfaced by the broker's proxy plane. Other engine-manager engines - * are not currently routed across nodes. - */ -export type ProxyEngine = Extract -export const PROXY_ENGINES: readonly ProxyEngine[] = ['ollama', 'lm-studio'] - -/** Map a proxy node source onto the engine it describes. */ -const PROXY_SOURCE_ENGINE: Record = { - 'ollama-proxy': 'ollama', - 'lmstudio-proxy': 'lm-studio' -} - /** Per-engine presence on a node — each proxy reports its own engine. */ interface EnginePresence { up: boolean @@ -186,10 +173,7 @@ function setEngine( engine: ProxyEngine, presence: EnginePresence ): Record { - return { - ollama: engine === 'ollama' ? presence : engines.ollama, - 'lm-studio': engine === 'lm-studio' ? presence : engines['lm-studio'] - } + return { ...engines, [engine]: presence } } /** @@ -401,11 +385,6 @@ export function parseWorkloadsInitial(value: JsonValue | undefined): Workload[] return workloads } -/** True for an engine fronted by a broker-supervised reverse proxy. */ -export function isProxyEngine(engine: EngineType): engine is ProxyEngine { - return engine === 'ollama' || engine === 'lm-studio' -} - const PENDING_OP_IDLE_TIMEOUT_MS = 90_000 // Vendor installers can be quiet for minutes; allow a bounded 30-minute window. const INSTALL_PENDING_OP_IDLE_TIMEOUT_MS = 30 * 60_000 @@ -721,7 +700,7 @@ function parseProxyNode(params: JsonValue | undefined, engine: ProxyEngine): Mod } return { id, - sources: [engine === 'ollama' ? 'ollama-proxy' : 'lmstudio-proxy'], + sources: [proxySourceForEngine(engine)], // `Node.Host` is the hostname; empty for the self-bridge manual node, // in which case the broker discovery entry supplies the display name on // merge (see mergeNode). Never fall back to the UUID id here. @@ -2274,12 +2253,9 @@ class ModularBridgeState { } handleNotification(notification: JsonRpcNotification): void { - if (notification.source === 'ollama-proxy') { - this.handleProxyNotification(notification, 'ollama') - return - } - if (notification.source === 'lmstudio-proxy') { - this.handleProxyNotification(notification, 'lm-studio') + const proxyEngine = proxyEngineFromSource(notification.source) + if (proxyEngine) { + this.handleProxyNotification(notification, proxyEngine) return } if (notification.source === 'broker') { @@ -2327,7 +2303,7 @@ class ModularBridgeState { if (notification.method === 'node/discovered' || notification.method === 'node/updated') { const node = parseProxyNode(notification.params, engine) if (!node) return - this.upsertNode(node, engine === 'ollama' ? 'ollama-proxy' : 'lmstudio-proxy') + this.upsertNode(node, proxySourceForEngine(engine)) } } @@ -2339,7 +2315,7 @@ class ModularBridgeState { private clearNodeEngine(nodeId: string, engine: ProxyEngine): void { const existing = this.nodes.get(nodeId) if (!existing) return - const source: BrokerNodeSource = engine === 'ollama' ? 'ollama-proxy' : 'lmstudio-proxy' + const source = proxySourceForEngine(engine) const sources = removeSource(existing.sources, source) if (sources.length === 0 && !existing.nodeInfoUp) { this.removeNodeEntry(nodeId) @@ -2589,7 +2565,7 @@ class ModularBridgeState { // A proxy source (ollama-proxy / lmstudio-proxy): refresh only that // engine's presence; keep the other engine, telemetry, and node-info. - const engine = PROXY_SOURCE_ENGINE[source] + const engine = proxyEngineFromSource(source) return { ...next, sources: mergeSources(existing.sources, source), diff --git a/desktop/src/electron/service-bridge/modular-supervisor.ts b/desktop/src/electron/service-bridge/modular-supervisor.ts index 8cd1012b..e5949b03 100644 --- a/desktop/src/electron/service-bridge/modular-supervisor.ts +++ b/desktop/src/electron/service-bridge/modular-supervisor.ts @@ -17,19 +17,24 @@ import getErrorString from '@/shared/utils/get-error-string' import { currentPlatform } from '@/shared/utils/platform' import { getModularBridgeState, - isProxyEngine, isUpstreamUnreachableError, parseServiceErrors, - parseWorkloadsInitial, + parseWorkloadsInitial +} from './modular-state' +import { + isProxyEngine, + proxyEngineFromManagerId, + proxyEngineFromSource, + proxySourceForEngine, PROXY_ENGINES, PROXY_NODE_SOURCES, type ProxyEngine -} from './modular-state' +} from './proxy-engines' import { emitBridgePush } from './broadcaster' import { parseEngineSettings } from './engine-settings' import { resolvePullCatchError } from './pull-error-handling' import { serviceLogLevel } from './service-log-level' -import { engineManagerName, engineTypeFromManagerName } from '@/shared/utils/engines' +import { engineManagerName } from '@/shared/utils/engines' import { isFirstRun } from '@/electron/config/ui-config' import { parseClusterNodes, parseInvite, parseNodeIdentity } from './cluster-json' import { startNodeInfoPoller, stopNodeInfoPoller } from './node-info-poller' @@ -292,17 +297,6 @@ function emptyLocalEngineBridge(): LocalEngineBridge { return { running: false, port: 0, bridgedId: '', bridgedPort: 0, selfWarned: false } } -/** Translate an engine-manager engine id into a proxy engine, or null. */ -function proxyEngineFromManagerId(id: string): ProxyEngine | null { - const engine = engineTypeFromManagerName(id) - return engine && isProxyEngine(engine) ? engine : null -} - -/** The broker relay namespace fronting an engine's reverse proxy. */ -function proxyRelayPrefix(engine: ProxyEngine): string { - return engine === 'ollama' ? 'ollama-proxy' : 'lmstudio-proxy' -} - /** * Spawns and supervises the modular backend. * @@ -677,7 +671,7 @@ class ModularSupervisor { method: string, params?: JsonValue ): Promise { - return this.callProcess('broker', `${proxyRelayPrefix(engine)}:${method}`, params) + return this.callProcess('broker', `${proxySourceForEngine(engine)}:${method}`, params) } /** @@ -873,8 +867,9 @@ class ModularSupervisor { } } await subscribe('discovery:subscribe', 'subscribe to broker discovery') - await subscribe('ollama-proxy:subscribe', 'subscribe to broker ollama-proxy relay') - await subscribe('lmstudio-proxy:subscribe', 'subscribe to broker lmstudio-proxy relay') + for (const source of PROXY_NODE_SOURCES) { + await subscribe(`${source}:subscribe`, `subscribe to broker ${source} relay`) + } // Engine events are opt-in and replay no baseline — subscribe then hydrate. await subscribe('engine:subscribe', 'subscribe to broker engine relay') await subscribe('workloads:subscribe', 'subscribe to broker workloads stream') @@ -1071,12 +1066,12 @@ class ModularSupervisor { try { const result = await this.callProcess( 'broker', - `${proxyRelayPrefix(engine)}:get-status` + `${proxySourceForEngine(engine)}:get-status` ) const obj = objectValue(result) if (obj && booleanValue(obj.ready)) { getModularBridgeState().handleNotification({ - source: engine === 'ollama' ? 'ollama-proxy' : 'lmstudio-proxy', + source: proxySourceForEngine(engine), method: 'ready', params: { port: numberValue(obj.port) } }) @@ -1097,14 +1092,14 @@ class ModularSupervisor { if (!obj || !Array.isArray(obj.nodes)) return for (const node of obj.nodes) { getModularBridgeState().handleNotification({ - source: engine === 'ollama' ? 'ollama-proxy' : 'lmstudio-proxy', + source: proxySourceForEngine(engine), method: 'node/discovered', params: node }) } } catch (err) { log.verbose({ - sublevel: proxyRelayPrefix(engine), + sublevel: proxySourceForEngine(engine), message: `Unable to hydrate ${engine} proxy nodes: ${getErrorString(err)}` }) } @@ -1263,12 +1258,7 @@ class ModularSupervisor { this.scheduleRemoteEngineStatusRefresh() } - const proxyEngine: ProxyEngine | null = - event.source === 'ollama-proxy' - ? 'ollama' - : event.source === 'lmstudio-proxy' - ? 'lm-studio' - : null + const proxyEngine = proxyEngineFromSource(event.source) if (proxyEngine && event.method === 'ready') { // A (re)bound proxy starts with an empty manual-node set, so forget // what we think we bridged and re-push the local node if applicable. @@ -2187,7 +2177,7 @@ class ModularSupervisor { if (!bridge.selfWarned) { bridge.selfWarned = true log.warn({ - sublevel: proxyRelayPrefix(engine), + sublevel: proxySourceForEngine(engine), message: `Skipping local-node ${engine} proxy bridge: engine port ` + `${bridge.port} matches the proxy's own listen port ` + @@ -2214,7 +2204,7 @@ class ModularSupervisor { bridge.bridgedPort = bridge.port } catch (err) { log.warn({ - sublevel: proxyRelayPrefix(engine), + sublevel: proxySourceForEngine(engine), message: `Failed to bridge local node into ${engine} proxy: ${getErrorString(err)}` }) } @@ -2229,7 +2219,7 @@ class ModularSupervisor { await this.callProxy(engine, 'node/remove-manual', { id: previousId }) } catch (err) { log.verbose({ - sublevel: proxyRelayPrefix(engine), + sublevel: proxySourceForEngine(engine), message: `Local node was not bridged into ${engine} proxy: ${getErrorString(err)}` }) } diff --git a/desktop/src/electron/service-bridge/proxy-engines.ts b/desktop/src/electron/service-bridge/proxy-engines.ts new file mode 100644 index 00000000..f72d42ad --- /dev/null +++ b/desktop/src/electron/service-bridge/proxy-engines.ts @@ -0,0 +1,58 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +import type { EngineManagerName, EngineType } from '@/shared/types/engines' + +/** + * Engines currently consumed through the broker's proxy plane. Keep this set + * separate from the complete engine registry so an engine can be declared + * before every proxy-state consumer is ready to handle it. + */ +export type ProxyEngine = Extract +export type ProxyNodeSource = 'ollama-proxy' | 'lmstudio-proxy' + +interface ProxyEngineIdentity { + managerName: EngineManagerName + source: ProxyNodeSource +} + +const PROXY_ENGINE_IDENTITIES: Record = { + ollama: { + managerName: 'ollama', + source: 'ollama-proxy' + }, + 'lm-studio': { + managerName: 'lmstudio', + source: 'lmstudio-proxy' + } +} + +export const PROXY_ENGINES: readonly ProxyEngine[] = ['ollama', 'lm-studio'] + +export const PROXY_NODE_SOURCES: readonly ProxyNodeSource[] = PROXY_ENGINES.map( + engine => PROXY_ENGINE_IDENTITIES[engine].source +) + +export function isProxyEngine(engine: EngineType): engine is ProxyEngine { + return PROXY_ENGINES.some(candidate => candidate === engine) +} + +export function proxySourceForEngine(engine: ProxyEngine): ProxyNodeSource { + return PROXY_ENGINE_IDENTITIES[engine].source +} + +export function proxyEngineFromSource(source: ProxyNodeSource): ProxyEngine +export function proxyEngineFromSource(source: string): ProxyEngine | null +export function proxyEngineFromSource(source: string): ProxyEngine | null { + for (const engine of PROXY_ENGINES) { + if (PROXY_ENGINE_IDENTITIES[engine].source === source) return engine + } + return null +} + +export function proxyEngineFromManagerId(managerName: string): ProxyEngine | null { + for (const engine of PROXY_ENGINES) { + if (PROXY_ENGINE_IDENTITIES[engine].managerName === managerName) return engine + } + return null +} diff --git a/desktop/src/shared/constants/modular-binaries.ts b/desktop/src/shared/constants/modular-binaries.ts index cc372104..e6d83d5a 100644 --- a/desktop/src/shared/constants/modular-binaries.ts +++ b/desktop/src/shared/constants/modular-binaries.ts @@ -65,7 +65,7 @@ export const MODULAR_RUNTIME_BINARIES: ModularRuntimeBinary[] = [ // naming it after an engine would claim a per-engine process that does // not exist. // The per-engine identities are the relay sources (`ollama-proxy` / - // `lmstudio-proxy`), which live in modular-state.ts. + // `lmstudio-proxy`), which live in proxy-engines.ts. processName: 'nvpair-proxy', baseName: 'nvpair-proxy', args: [], diff --git a/desktop/tests/modular/proxy-engine-identity.test.ts b/desktop/tests/modular/proxy-engine-identity.test.ts new file mode 100644 index 00000000..f6354b3f --- /dev/null +++ b/desktop/tests/modular/proxy-engine-identity.test.ts @@ -0,0 +1,84 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +import { describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ BrowserWindow: { getAllWindows: () => [] } })) +vi.mock('@/electron/window', () => ({ createOverviewWindow: vi.fn() })) + +import { getModularBridgeState } from '@/electron/service-bridge/modular-state' +import { + isProxyEngine, + PROXY_ENGINES, + PROXY_NODE_SOURCES, + proxyEngineFromManagerId, + proxyEngineFromSource, + proxySourceForEngine +} from '@/electron/service-bridge/proxy-engines' +import { engineManagerName } from '@/shared/utils/engines' + +describe('proxy engine identity', () => { + it('keeps the enabled proxy set and ordering unchanged', () => { + expect(PROXY_ENGINES).toEqual(['ollama', 'lm-studio']) + expect(PROXY_NODE_SOURCES).toEqual(['ollama-proxy', 'lmstudio-proxy']) + expect(isProxyEngine('ollama')).toBe(true) + expect(isProxyEngine('lm-studio')).toBe(true) + expect(isProxyEngine('llama-cpp')).toBe(false) + }) + + it('round-trips engine, manager, and relay source identities', () => { + for (const engine of PROXY_ENGINES) { + const source = proxySourceForEngine(engine) + expect(proxyEngineFromSource(source)).toBe(engine) + expect(proxyEngineFromManagerId(engineManagerName(engine))).toBe(engine) + } + expect(proxyEngineFromSource('other-proxy')).toBeNull() + expect(proxyEngineFromManagerId('other')).toBeNull() + expect(proxyEngineFromManagerId('llamacpp')).toBeNull() + }) + + it('routes each proxy notification to the mapped engine', () => { + const state = getModularBridgeState() + const nodeId = 'proxy-identity-remote' + state.setSelfId('proxy-identity-self') + + state.handleNotification({ + source: 'ollama-proxy', + method: 'node/discovered', + params: { + id: nodeId, + host: 'proxy-identity-host', + port: 11434, + addresses: ['192.0.2.40'], + ip: '192.0.2.40' + } + }) + state.handleNotification({ + source: 'lmstudio-proxy', + method: 'node/discovered', + params: { + id: nodeId, + host: 'proxy-identity-host', + port: 1234, + addresses: ['192.0.2.40'], + ip: '192.0.2.40' + } + }) + + const statuses = state + .getEngineInitialState() + .statuses.filter(status => status.nodeId === nodeId) + expect(statuses).toEqual([ + expect.objectContaining({ + engineType: 'ollama', + processStatus: 'running', + proxyPort: 11434 + }), + expect.objectContaining({ + engineType: 'lm-studio', + processStatus: 'running', + proxyPort: 1234 + }) + ]) + }) +}) From a569b986919791c587022b0dd71006b16cc7af36 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 16:21:27 -0700 Subject: [PATCH 16/70] feat(desktop): bridge llama.cpp service state Signed-off-by: Sherief Farouk --- desktop/docs/service-contract-exceptions.json | 1 + .../electron/service-bridge/empty-handlers.ts | 7 +- .../electron/service-bridge/modular-state.ts | 41 +++--- .../service-bridge/modular-supervisor.ts | 65 ++++++---- .../electron/service-bridge/proxy-engines.ts | 10 +- .../tests/modular/engine-command-load.test.ts | 38 ++++++ .../modular/lmstudio-stale-model.test.ts | 31 ++++- .../modular/proxy-engine-identity.test.ts | 121 +++++++++++++++++- 8 files changed, 261 insertions(+), 53 deletions(-) diff --git a/desktop/docs/service-contract-exceptions.json b/desktop/docs/service-contract-exceptions.json index 90ae4eec..bb6d4e79 100644 --- a/desktop/docs/service-contract-exceptions.json +++ b/desktop/docs/service-contract-exceptions.json @@ -9,6 +9,7 @@ "engine:restore-enabled": "Broker-internal startup restoration. nvpair-ui-broker emits engine:restore-enabled directly to its supervised engine-manager after the managed Ollama port gate and on manager respawn; it is not a renderer/UI notification.", "ollama-proxy:ready": "Consumed, not missing: the broker relays it and normalizeBrokerProxy (modular-supervisor.ts) strips the `ollama-proxy:` prefix, so the bridge handles the de-prefixed `ready` (sets proxyPort). The literal `ollama-proxy:ready` is intentionally absent from our TS — extractor limitation, not a gap.", "lmstudio-proxy:ready": "Consumed, not missing: the LM Studio counterpart of ollama-proxy:ready, de-prefixed by the same normalizeBrokerProxy loop. It only became visible to the checker when METHOD_RE started accepting hyphens in a namespace segment; before that the whole lmstudio-proxy:* surface was silently unmatched.", + "llamacpp-proxy:ready": "Consumed, not missing: the llama.cpp counterpart of ollama-proxy:ready, de-prefixed by the same normalizeBrokerProxy loop and mapped to the desktop llama-cpp engine identity.", "node/selection-changed": "Automatic routing has no selected-node UI, so PAIR deliberately does not consume proxy selection changes.", "proxy/request": "Per-request proxy telemetry is not rendered; workload lifecycle uses the broker workloads stream.", "proxy/request-started": "Per-request proxy telemetry is not rendered; see proxy/request.", diff --git a/desktop/src/electron/service-bridge/empty-handlers.ts b/desktop/src/electron/service-bridge/empty-handlers.ts index 14fd3d89..6dd3fe76 100644 --- a/desktop/src/electron/service-bridge/empty-handlers.ts +++ b/desktop/src/electron/service-bridge/empty-handlers.ts @@ -260,8 +260,8 @@ function routeEngineManagerCommand(payload: WsInvokeRequest<'engine:command'>): // first-class load action, so we POST its `run_model` HTTP action // (`/api/generate`) with no prompt: Ollama loads the model into VRAM // and returns immediately (`done_reason: "load"`) without generating. - // LM Studio declares a real `load_model` CLI action (`lms load`). - // Both require the engine running (HTTP/CLI action), matching the + // LM Studio and llama.cpp declare real `load_model` manifest actions. + // All require the engine running (HTTP/CLI action), matching the // disabled rule in ModelRow.tsx. // See docs/services-parity.md#models. if (payload.model) { @@ -401,8 +401,7 @@ async function handleEngineCommand(payload?: WsInvokeRequest<'engine:command'>): // Proxy node selection is owned by the backend nvpair-job-scheduler (it drives // node/set-priority via the broker); PAIR UI never pins node/select, and a user // clicking "Load" must not pin a route. Ollama loads via its `run_model` HTTP - // action, LM Studio via its `load_model` CLI action — both handled locally in - // routeEngineManagerCommand. + // action; LM Studio and llama.cpp use their manifest `load_model` actions. // Remote peers: `nvpair-engine-manager` exposes `engine:remote-*` client methods // (cluster mTLS `ec` surface) for install/start/stop/pull and model ops. diff --git a/desktop/src/electron/service-bridge/modular-state.ts b/desktop/src/electron/service-bridge/modular-state.ts index eef2ad26..4463803f 100644 --- a/desktop/src/electron/service-bridge/modular-state.ts +++ b/desktop/src/electron/service-bridge/modular-state.ts @@ -139,7 +139,8 @@ interface ModularNode { // fallback path (a peer that sends no per-engine attribution). models: string[] // Per-engine attribution of the enriched model list, keyed by engine-manager - // engine name ("ollama", "lmstudio"), carried on `AvailableNode.modelsByEngine`. + // engine name ("ollama", "lmstudio", "llamacpp"), carried on + // `AvailableNode.modelsByEngine`. // When present it is authoritative — {@link toEngineModels} attributes each // model to the engine that actually serves it, so a dual-engine node no // longer blanks its cards. A present engine key with [] is known empty; an @@ -148,7 +149,7 @@ interface ModularNode { // noderec.DirectoryNode.EngineModels. modelsByEngine: Record // Per-engine set of models currently loaded in memory, keyed by - // engine-manager engine name ("ollama", "lmstudio"), carried on + // engine-manager engine name ("ollama", "lmstudio", "llamacpp"), carried on // `AvailableNode.loadedByEngine`. Normally a subset of // {@link modelsByEngine}. An engine key with an empty list means // "running, nothing loaded"; a missing key means loaded state wasn't @@ -164,7 +165,11 @@ function emptyPresence(): EnginePresence { } function emptyEngines(): Record { - return { ollama: emptyPresence(), 'lm-studio': emptyPresence() } + return { + ollama: emptyPresence(), + 'lm-studio': emptyPresence(), + 'llama-cpp': emptyPresence() + } } /** Immutably set one engine's presence, preserving the other. */ @@ -847,7 +852,7 @@ function sameNode(left: ModularNode, right: ModularNode): boolean { /** * Compare two per-engine model maps. The attribution can change while the flat - * union stays identical (e.g. a model that both engines now serve moved between + * union stays identical (e.g. a model that multiple engines now serve moved between * the per-engine buckets), so this must be checked independently of * {@link sameStringList} on `models` or a card would miss the re-attribution. */ @@ -871,8 +876,12 @@ class ModularBridgeState { private logs: LogEntry[] = [] // Per-engine bound proxy port reported by the broker. 0 = not reported yet; // we never fabricate a default — an unknown port surfaces as null, not a - // guess. `ollama` is the `ollama-proxy`, `lm-studio` is the `lmstudio-proxy`. - private proxyPorts: Record = { ollama: 0, 'lm-studio': 0 } + // guess. Keys map to relay sources through proxy-engines.ts. + private proxyPorts: Record = { + ollama: 0, + 'lm-studio': 0, + 'llama-cpp': 0 + } private selfId: string | null = null /** * Authoritative local-engine facts from `nvpair-engine-manager`, keyed by @@ -1918,8 +1927,8 @@ class ModularBridgeState { * Remote peers prefer facts over proxy presence for attribution. Presence is * per-engine but coarse — it only says the peer is reachable for that engine, * not which engine actually serves a given model. For a pre-attribution peer - * that sends no `modelsByEngine`, whenever both engines are momentarily `up`, - * leaning on presence makes two engines look "active" at once, which defeats + * that sends no `modelsByEngine`, whenever multiple engines are momentarily `up`, + * leaning on presence makes multiple engines look "active" at once, which defeats * the single-active-engine attribution in {@link modelsForEngine} and blanks * the model list. So once a peer's authoritative facts exist they are the sole * truth; presence is used only when facts are absent (an unclustered peer, or @@ -1945,7 +1954,7 @@ class ModularBridgeState { * so a dual-engine node attributes each model to the engine that serves it. * A pre-attribution peer sends no map, so the flat `AvailableNode.models` * union is attributed only when exactly one proxy engine is active (never - * cross-attributed when both run). Mirrors + * cross-attributed when multiple run). Mirrors * noderec.DirectoryNode.EngineModels. */ private modelsForEngine(node: ModularNode, engine: ProxyEngine): string[] { @@ -1959,7 +1968,7 @@ class ModularBridgeState { } // Fallback for a pre-attribution peer: the flat union is unattributed, so // attribute it only when exactly one proxy engine is active on the node, - // and never cross-attribute when both run. + // and never cross-attribute when multiple run. const active = PROXY_ENGINES.filter(candidate => this.engineActiveForModels(node, candidate) ) @@ -2061,14 +2070,14 @@ class ModularBridgeState { ? 'stopped' : 'not-installed', // The engine-manager reports the manifest's configured port for a - // pinned-port engine (Ollama 11434, managed LM Studio 1235 behind - // its 1234 proxy facade) whether it is running or merely installed, + // pinned-port engine (for example managed llama.cpp on 8081 behind + // its 8080 proxy facade) whether it is running or merely installed, // so surface it in both states — // matching the `EngineStatusData.enginePort` contract. Auto-assign // engines report 0 until started, which stays null. enginePort: facts.installed && facts.port > 0 ? facts.port : null, // Each proxy-fronted engine has its own broker proxy - // (`ollama-proxy` / `lmstudio-proxy`); report that engine's bound + // (mapped in proxy-engines.ts); report that engine's bound // proxy port. Loopback-only engines get null. proxyPort: isProxyEngine(engineType) ? this.getProxyPort(engineType) : null } @@ -2508,7 +2517,7 @@ class ModularBridgeState { // install/running state; discovery only fills in models. A remote node // has no local engine-manager, so its status comes from authoritative // peer facts or its advertisement, and is omitted when neither is known. - // Push per proxy-engine (Ollama + LM Studio) so both light up per node. + // Push per proxy engine so each one lights up independently per node. const isSelf = merged.id === this.selfId for (const engine of PROXY_ENGINES) { if (!isSelf) { @@ -2563,8 +2572,8 @@ class ModularBridgeState { } } - // A proxy source (ollama-proxy / lmstudio-proxy): refresh only that - // engine's presence; keep the other engine, telemetry, and node-info. + // A proxy source refreshes only that engine's presence; keep the other + // engines, telemetry, and node-info. const engine = proxyEngineFromSource(source) return { ...next, diff --git a/desktop/src/electron/service-bridge/modular-supervisor.ts b/desktop/src/electron/service-bridge/modular-supervisor.ts index e5949b03..12001ce6 100644 --- a/desktop/src/electron/service-bridge/modular-supervisor.ts +++ b/desktop/src/electron/service-bridge/modular-supervisor.ts @@ -97,7 +97,7 @@ function getModularBinaryPath(baseName: string): string { * Read the build provenance (`sourceFingerprint` + `services`) and per-component * versions that `scripts/build-modular-binaries.ts` stamps into * `cli-bin/manifest.json`. - * `components` is keyed by binary base name (e.g. `ollama-proxy`). Returns empty + * `components` is keyed by binary base name (e.g. `nvpair-proxy`). Returns empty * values when the manifest is absent. */ export function readCliBinManifest(): { @@ -160,26 +160,43 @@ function booleanValue(value: JsonValue | undefined): boolean { * Extract model names from a `nvpair-engine-manager` `list_models` action result. * The action returns the engine's raw response, which differs per engine: * Ollama's `/api/tags` yields `{ models: [{ name }] }`, LM Studio's native - * `/api/v1/models` yields `{ models: [{ key }] }`. A present empty array is - * authoritative; a missing or malformed inventory throws so callers retain or - * fall back to their last-good source instead of silently clearing it. + * `/api/v1/models` yields `{ models: [{ key }] }`, and llama.cpp's router + * `/models` yields `{ data: [{ id }] }`. A present empty array is authoritative; + * a missing or malformed inventory throws so callers retain or fall back to + * their last-good source instead of silently clearing it. */ export function parseListModelNames(result: JsonValue | undefined): string[] { const obj = objectValue(result) if (!obj) throw new Error('list_models returned a non-object response') - const names: string[] = [] + + let rows: JsonValue[] + let fields: string[] if (Array.isArray(obj.models)) { - for (const entry of obj.models) { - const row = objectValue(entry) - const name = stringValue(row?.name) || stringValue(row?.key) - if (name) names.push(name) - } - if (obj.models.length > 0 && names.length === 0) { - throw new Error('list_models returned no usable model names') + rows = obj.models + fields = ['name', 'key'] + } else if (obj.models !== undefined) { + throw new Error('list_models response is missing its model array') + } else if (Array.isArray(obj.data)) { + rows = obj.data + fields = ['id'] + } else { + throw new Error('list_models response is missing its model array') + } + + const names: string[] = [] + for (const entry of rows) { + const row = objectValue(entry) + for (const field of fields) { + const name = stringValue(row?.[field]) + if (!name) continue + names.push(name) + break } - return names } - throw new Error('list_models response is missing its model array') + if (rows.length > 0 && names.length === 0) { + throw new Error('list_models returned no usable model names') + } + return names } /** @@ -209,16 +226,16 @@ function normalizeLogLevel(value: string | undefined): ModularLogLevel { /** * Shape the `pull_model` action params per engine. Ollama's `pull_model` body is - * sent verbatim to `/api/pull` (reads `name`); LM Studio's CLI action templates - * `{model}` into `lms get {model} --yes`. Sending the wrong key leaves the - * placeholder unresolved and the engine-manager rejects the call. + * sent verbatim to `/api/pull` (reads `name`); every other manifest consumes + * `model` either as an HTTP body field or a CLI template. Sending the wrong key + * leaves the placeholder unresolved or fails body-schema validation. */ -function pullModelParams(engineManagerEngine: string, model: string): JsonObject { - return engineManagerEngine === 'lmstudio' ? { model } : { name: model } +export function pullModelParams(engineManagerEngine: string, model: string): JsonObject { + return engineManagerEngine === 'ollama' ? { name: model } : { model } } function deleteModelParams(engineManagerEngine: string, model: string): JsonObject { - return engineManagerEngine === 'lmstudio' ? { model } : { name: model } + return engineManagerEngine === 'ollama' ? { name: model } : { model } } /** @@ -304,12 +321,12 @@ function emptyLocalEngineBridge(): LocalEngineBridge { * `docs/services-backend.md`): * * - The `nvpair-ui-broker` is the **only** Electron-spawned binary and is itself the - * parent of every broker-owned worker (`ollama-proxy`, `lmstudio-proxy`, + * parent of every broker-owned worker (`nvpair-proxy`, * `nvpair-node-scanner`, `nvpair-node-info`, `nvpair-workload-manager`, * `nvpair-cluster-manager`, `nvpair-node-settings`, `nvpair-manual-nodes`, * `nvpair-engine-manager`, `nvpair-errors`, `nvpair-job-scheduler`). Electron passes their resolved paths to * the broker (see `brokerStartupArgs`) and reaches each through a broker relay: - * `ollama-proxy:` / `lmstudio-proxy:` for the two engine proxies, `engine:` for the + * the mapped `-proxy:` facade relays, `engine:` for the * engine-manager, `errors:` for the error pipeline, `node/*` for manual nodes, * `settings/*` and `cluster:` for the rest. Local inference jobs arrive on the * broker's `workloads:subscribe` stream. @@ -2126,8 +2143,8 @@ class ModularSupervisor { * engine:state-changed and reconcile. The engine:state-changed carries the * **real** local engine port, which is the one the proxy must route to (mDNS * self-discovery can advertise the wrong port even when it works). Applies to - * every proxy-fronted engine (Ollama → ollama-proxy, LM Studio → - * lmstudio-proxy); loopback-only engines are ignored. + * every proxy-fronted engine through the mapping in proxy-engines.ts; + * loopback-only engines are ignored. */ private updateLocalNodeBridgeFromEngineState(params: JsonValue | undefined): void { const obj = objectValue(params) diff --git a/desktop/src/electron/service-bridge/proxy-engines.ts b/desktop/src/electron/service-bridge/proxy-engines.ts index f72d42ad..447246af 100644 --- a/desktop/src/electron/service-bridge/proxy-engines.ts +++ b/desktop/src/electron/service-bridge/proxy-engines.ts @@ -8,8 +8,8 @@ import type { EngineManagerName, EngineType } from '@/shared/types/engines' * separate from the complete engine registry so an engine can be declared * before every proxy-state consumer is ready to handle it. */ -export type ProxyEngine = Extract -export type ProxyNodeSource = 'ollama-proxy' | 'lmstudio-proxy' +export type ProxyEngine = Extract +export type ProxyNodeSource = 'ollama-proxy' | 'lmstudio-proxy' | 'llamacpp-proxy' interface ProxyEngineIdentity { managerName: EngineManagerName @@ -24,10 +24,14 @@ const PROXY_ENGINE_IDENTITIES: Record = { 'lm-studio': { managerName: 'lmstudio', source: 'lmstudio-proxy' + }, + 'llama-cpp': { + managerName: 'llamacpp', + source: 'llamacpp-proxy' } } -export const PROXY_ENGINES: readonly ProxyEngine[] = ['ollama', 'lm-studio'] +export const PROXY_ENGINES: readonly ProxyEngine[] = ['ollama', 'lm-studio', 'llama-cpp'] export const PROXY_NODE_SOURCES: readonly ProxyNodeSource[] = PROXY_ENGINES.map( engine => PROXY_ENGINE_IDENTITIES[engine].source diff --git a/desktop/tests/modular/engine-command-load.test.ts b/desktop/tests/modular/engine-command-load.test.ts index 1689cc1d..bf379fef 100644 --- a/desktop/tests/modular/engine-command-load.test.ts +++ b/desktop/tests/modular/engine-command-load.test.ts @@ -91,4 +91,42 @@ describe('local model load command', () => { expect.any(Function) ) }) + + it('routes llama.cpp load and unload through its manifest actions', async () => { + await handleServiceBridgeInvoke('engine:command', { + command: 'loadModel', + engineType: 'llama-cpp', + nodeId: 'local-node', + model: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M' + }) + await handleServiceBridgeInvoke('engine:command', { + command: 'unloadModel', + engineType: 'llama-cpp', + nodeId: 'local-node', + model: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M' + }) + + expect(mocks.supervisor.sendProcess).toHaveBeenNthCalledWith( + 1, + 'broker', + 'engine:action', + { + engine: 'llamacpp', + action: 'load_model', + params: { model: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M' } + }, + expect.any(Function) + ) + expect(mocks.supervisor.sendProcess).toHaveBeenNthCalledWith( + 2, + 'broker', + 'engine:action', + { + engine: 'llamacpp', + action: 'unload_model', + params: { model: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M' } + }, + expect.any(Function) + ) + }) }) diff --git a/desktop/tests/modular/lmstudio-stale-model.test.ts b/desktop/tests/modular/lmstudio-stale-model.test.ts index 8b37e862..23a79fa9 100644 --- a/desktop/tests/modular/lmstudio-stale-model.test.ts +++ b/desktop/tests/modular/lmstudio-stale-model.test.ts @@ -12,9 +12,38 @@ vi.mock('@/electron/window', () => ({ createOverviewWindow: vi.fn() })) import { getModularBridgeState } from '@/electron/service-bridge/modular-state' import { getModularSupervisor, - parseListModelNames + parseListModelNames, + pullModelParams } from '@/electron/service-bridge/modular-supervisor' +describe('engine model wire formats', () => { + it('parses strict llama.cpp router inventories', () => { + expect( + parseListModelNames({ + data: [ + { id: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M', status: { value: 'loaded' } }, + { id: 'bartowski/Qwen2.5-3B-Instruct-GGUF:Q4_K_M' } + ] + }) + ).toEqual([ + 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M', + 'bartowski/Qwen2.5-3B-Instruct-GGUF:Q4_K_M' + ]) + expect(parseListModelNames({ data: [] })).toEqual([]) + expect(() => parseListModelNames({ data: null })).toThrow('missing its model array') + expect(() => parseListModelNames({ data: [{}] })).toThrow('no usable model names') + expect(() => + parseListModelNames({ models: null, data: [{ id: 'must-not-mask-malformed-models' }] }) + ).toThrow('missing its model array') + }) + + it('uses the manifest model field for llama.cpp pulls', () => { + expect(pullModelParams('ollama', 'demo')).toEqual({ name: 'demo' }) + expect(pullModelParams('lmstudio', 'demo')).toEqual({ model: 'demo' }) + expect(pullModelParams('llamacpp', 'demo')).toEqual({ model: 'demo' }) + }) +}) + describe('LM Studio model reconciliation', () => { it('parses the native inventory and distinguishes explicit empty from unknown', () => { expect( diff --git a/desktop/tests/modular/proxy-engine-identity.test.ts b/desktop/tests/modular/proxy-engine-identity.test.ts index f6354b3f..69031984 100644 --- a/desktop/tests/modular/proxy-engine-identity.test.ts +++ b/desktop/tests/modular/proxy-engine-identity.test.ts @@ -18,12 +18,12 @@ import { import { engineManagerName } from '@/shared/utils/engines' describe('proxy engine identity', () => { - it('keeps the enabled proxy set and ordering unchanged', () => { - expect(PROXY_ENGINES).toEqual(['ollama', 'lm-studio']) - expect(PROXY_NODE_SOURCES).toEqual(['ollama-proxy', 'lmstudio-proxy']) + it('declares the complete proxy set in stable order', () => { + expect(PROXY_ENGINES).toEqual(['ollama', 'lm-studio', 'llama-cpp']) + expect(PROXY_NODE_SOURCES).toEqual(['ollama-proxy', 'lmstudio-proxy', 'llamacpp-proxy']) expect(isProxyEngine('ollama')).toBe(true) expect(isProxyEngine('lm-studio')).toBe(true) - expect(isProxyEngine('llama-cpp')).toBe(false) + expect(isProxyEngine('llama-cpp')).toBe(true) }) it('round-trips engine, manager, and relay source identities', () => { @@ -34,7 +34,7 @@ describe('proxy engine identity', () => { } expect(proxyEngineFromSource('other-proxy')).toBeNull() expect(proxyEngineFromManagerId('other')).toBeNull() - expect(proxyEngineFromManagerId('llamacpp')).toBeNull() + expect(proxyEngineFromManagerId('llamacpp')).toBe('llama-cpp') }) it('routes each proxy notification to the mapped engine', () => { @@ -64,6 +64,52 @@ describe('proxy engine identity', () => { ip: '192.0.2.40' } }) + state.handleNotification({ + source: 'llamacpp-proxy', + method: 'ready', + params: { port: 8080 } + }) + state.handleNotification({ + source: 'llamacpp-proxy', + method: 'node/discovered', + params: { + id: nodeId, + host: 'proxy-identity-host', + port: 8080, + addresses: ['192.0.2.40'], + ip: '192.0.2.40' + } + }) + state.handleNotification({ + source: 'broker', + method: 'discovery:nodes-changed', + params: { + nodes: [ + { + hostUuid: nodeId, + name: 'proxy-identity-host', + ipAddress: '192.0.2.40', + port: 14318, + models: ['owner/alpha:Q4_K_M', 'owner/beta:Q4_K_M'], + modelsByEngine: { + llamacpp: ['owner/alpha:Q4_K_M', 'owner/beta:Q4_K_M'] + }, + loadedByEngine: { llamacpp: ['owner/alpha:Q4_K_M'] } + } + ] + } + }) + state.applyRemoteEngineFacts(nodeId, { + engines: [ + { + engine: 'llamacpp', + installed: true, + running: true, + healthy: true, + port: 8081 + } + ] + }) const statuses = state .getEngineInitialState() @@ -78,7 +124,72 @@ describe('proxy engine identity', () => { engineType: 'lm-studio', processStatus: 'running', proxyPort: 1234 + }), + expect.objectContaining({ + engineType: 'llama-cpp', + processStatus: 'running', + enginePort: 8081, + proxyPort: 8080 }) ]) + + const llamaModels = state + .getEngineInitialState() + .models.find(models => models.nodeId === nodeId && models.engineType === 'llama-cpp') + expect(llamaModels?.models).toEqual([ + expect.objectContaining({ name: 'owner/alpha:Q4_K_M', status: 'loaded' }), + expect.objectContaining({ name: 'owner/beta:Q4_K_M', status: 'idle' }) + ]) + }) + + it('projects local llama.cpp lifecycle and residency into the initial snapshot', () => { + const state = getModularBridgeState() + const nodeId = 'proxy-identity-local' + const model = 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M' + state.setSelfId(nodeId) + state.handleNotification({ + source: 'broker', + method: 'discovery:nodes-changed', + params: { + nodes: [ + { + hostUuid: nodeId, + name: 'proxy-identity-local-host', + ipAddress: '127.0.0.1', + port: 14318 + } + ] + } + }) + state.handleNotification({ + source: 'llamacpp-proxy', + method: 'ready', + params: { port: 8080 } + }) + state.applyEngineManagerStatus({ + engine: 'llamacpp', + installed: true, + running: true, + healthy: true, + port: 8081 + }) + state.setLocalEngineModels('llama-cpp', [model]) + state.applyLocalLoadedModels({ llamacpp: [model] }) + + const initial = state.getEngineInitialState() + expect( + initial.statuses.find( + status => status.nodeId === nodeId && status.engineType === 'llama-cpp' + ) + ).toMatchObject({ + processStatus: 'running', + enginePort: 8081, + proxyPort: 8080 + }) + expect( + initial.models.find( + models => models.nodeId === nodeId && models.engineType === 'llama-cpp' + )?.models + ).toEqual([expect.objectContaining({ name: model, status: 'loaded' })]) }) }) From efadf8a28a1f5f01804f154c6ac7c59c4933d3f9 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 18:26:40 -0700 Subject: [PATCH 17/70] feat(model-hub): add locked llama.cpp catalog Signed-off-by: Sherief Farouk --- desktop/src/electron/model-hub/index.ts | 14 ++-- .../electron/model-hub/llamacpp-catalog.ts | 62 +++++++++++++++ desktop/src/shared/types/engine-api.ts | 10 +-- .../src/ui/utils/match-downloaded-model.ts | 7 +- .../tests/modular/llamacpp-catalog.test.ts | 79 +++++++++++++++++++ 5 files changed, 160 insertions(+), 12 deletions(-) create mode 100644 desktop/src/electron/model-hub/llamacpp-catalog.ts create mode 100644 desktop/tests/modular/llamacpp-catalog.test.ts diff --git a/desktop/src/electron/model-hub/index.ts b/desktop/src/electron/model-hub/index.ts index d1fda727..54e28f70 100644 --- a/desktop/src/electron/model-hub/index.ts +++ b/desktop/src/electron/model-hub/index.ts @@ -8,6 +8,7 @@ import { lmStudioCatalogCache, type LmStudioCatalogModel } from '@/electron/model-hub/lmstudio-catalog' +import { loadLlamaCppModels } from './llamacpp-catalog' function ollamaToHubModel(m: OllamaTagsModel): EngineHubModel { const base = m.name.includes(':') ? m.name.slice(0, m.name.indexOf(':')) : m.name @@ -40,10 +41,9 @@ function lmStudioToHubModel(m: LmStudioCatalogModel): EngineHubModel { } /** - * Serve an engine's model hub. Ollama is served from the committed, locked list - * (`ollama-models.json`), so it returns instantly with no network access. LM - * Studio still fetches its live `lmstudio-community` catalog and awaits a cold - * cache's initial load. Engines without a hub return empty. + * Serve an engine's model hub. Ollama and llama.cpp use committed locked lists, + * so they return instantly with no network access. LM Studio still fetches its + * live `lmstudio-community` catalog and awaits a cold cache's initial load. */ export async function getEngineHubModels(engineType: EngineType): Promise { switch (engineType) { @@ -52,6 +52,8 @@ export async function getEngineHubModels(engineType: EngineType): Promise { + const quantifierAt = model.id.lastIndexOf(':') + const repoId = quantifierAt > 0 ? model.id.slice(0, quantifierAt) : model.id + const authorAt = repoId.indexOf('/') + return { + ...model, + name: model.id, + author: authorAt > 0 ? repoId.slice(0, authorAt) : '', + url: `https://huggingface.co/${repoId}`, + downloads: 0, + likes: 0, + tags: ['gguf', 'text-generation', 'Q4_K_M'] + } +}) + +/** Return the locked llama.cpp starter catalog without network access. */ +export function loadLlamaCppModels(): EngineHubModel[] { + return LLAMA_CPP_MODELS +} diff --git a/desktop/src/shared/types/engine-api.ts b/desktop/src/shared/types/engine-api.ts index 9187e658..350a15a0 100644 --- a/desktop/src/shared/types/engine-api.ts +++ b/desktop/src/shared/types/engine-api.ts @@ -11,11 +11,11 @@ import type { EngineStatusData, EngineType } from '@/shared/types/engines' import type { EngineModels, EngineProgress, EngineUpdateAvailable } from '@/shared/types/engines' /** - * One normalized model row returned by an engine-owned hub source (Ollama - * library scrape, LM Studio community catalog). The Electron-main model-hub - * module normalizes each upstream registry into this shape so the renderer - * maps it to a display row without per-engine JSON parsing. `id`/`name` carry - * the pull-ready identifier the engine's `pull_model` action expects. + * One normalized model row returned by an engine-owned hub source. The + * Electron-main model-hub module normalizes locked catalogs and live registries + * into this shape so the renderer maps them without per-engine parsing. + * `id`/`name` carry the pull-ready identifier the engine's `pull_model` action + * expects. */ export interface EngineHubModel { id: string diff --git a/desktop/src/ui/utils/match-downloaded-model.ts b/desktop/src/ui/utils/match-downloaded-model.ts index 2edf8361..cdce8576 100644 --- a/desktop/src/ui/utils/match-downloaded-model.ts +++ b/desktop/src/ui/utils/match-downloaded-model.ts @@ -17,6 +17,8 @@ import type { ModelEntry } from '@/ui/types/model-hub' * prefix so every quantization collapses to "downloaded". * - LM Studio: the download is keyed by `pullKey = /` (or * `//`); the hub id is `/`. + * - llama.cpp: router inventory and catalog use the same complete + * `/:` id, so only exact ids match. */ type DownloadedMatcher = (hubEntry: ModelEntry, downloaded: ModelItem) => boolean @@ -36,9 +38,12 @@ const matchHfPullKeyOrName: DownloadedMatcher = (hubEntry, d) => { return false } +const matchExactName: DownloadedMatcher = (hubEntry, downloaded) => downloaded.name === hubEntry.id + const MATCHERS: Partial> = { ollama: matchOllama, - 'lm-studio': matchHfPullKeyOrName + 'lm-studio': matchHfPullKeyOrName, + 'llama-cpp': matchExactName } export function isHubEntryDownloaded( diff --git a/desktop/tests/modular/llamacpp-catalog.test.ts b/desktop/tests/modular/llamacpp-catalog.test.ts new file mode 100644 index 00000000..b1951232 --- /dev/null +++ b/desktop/tests/modular/llamacpp-catalog.test.ts @@ -0,0 +1,79 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +import { describe, expect, it } from 'vitest' +import { getEngineHubModels } from '@/electron/model-hub' +import { loadLlamaCppModels } from '@/electron/model-hub/llamacpp-catalog' +import { isHubEntryDownloaded } from '@/ui/utils/match-downloaded-model' +import type { ModelItem } from '@/ui/types/engine-info' +import type { ModelEntry } from '@/ui/types/model-hub' + +function modelEntry(id: string): ModelEntry { + return { + id, + name: id, + author: id.slice(0, id.indexOf('/')), + url: `https://huggingface.co/${id.slice(0, id.lastIndexOf(':'))}`, + updatedAt: new Date(0) + } +} + +function modelItem(name: string, downloaded = true): ModelItem { + return { + name, + size: 0, + downloaded, + status: 'idle', + parameterSize: '', + quantization: '', + family: '', + digest: '', + sizeVram: null, + expiresAt: null, + expiry: '10m', + capabilities: [] + } +} + +describe('locked llama.cpp catalog', () => { + it('contains a small, unique, reviewed set of pull-ready models', () => { + const models = loadLlamaCppModels() + const publishers = new Set() + + expect(models.length).toBeGreaterThanOrEqual(3) + expect(models.length).toBeLessThanOrEqual(5) + expect(new Set(models.map(model => model.id)).size).toBe(models.length) + + for (const model of models) { + const repoId = model.id.slice(0, model.id.lastIndexOf(':')) + publishers.add(model.author) + expect(model.id).toMatch(/^(ggml-org|bartowski|unsloth)\/[^/:]+:Q4_K_M$/) + expect(model.name).toBe(model.id) + expect(model.author).toBe(repoId.slice(0, repoId.indexOf('/'))) + expect(model.url).toBe(`https://huggingface.co/${repoId}`) + expect(model.tags).toContain('text-generation') + expect(model.tags).toContain('Q4_K_M') + expect(Number.isNaN(Date.parse(model.updatedAt))).toBe(false) + expect(model.family).toBeTruthy() + expect(model.parameterSize).toBeTruthy() + } + expect(publishers).toEqual(new Set(['ggml-org', 'bartowski', 'unsloth'])) + }) + + it('serves the locked list through the engine hub without transformation', async () => { + await expect(getEngineHubModels('llama-cpp')).resolves.toEqual({ + models: loadLlamaCppModels() + }) + }) + + it('hides only an exact downloaded router model id', () => { + const id = loadLlamaCppModels()[0].id + const entry = modelEntry(id) + + expect(isHubEntryDownloaded('llama-cpp', entry, [modelItem(id)])).toBe(true) + expect( + isHubEntryDownloaded('llama-cpp', entry, [modelItem(id.slice(0, id.lastIndexOf(':')))]) + ).toBe(false) + expect(isHubEntryDownloaded('llama-cpp', entry, [modelItem(id, false)])).toBe(false) + }) +}) From f6be4da9f0f35ca6866cad147e7bd6190dc25278 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 18:44:29 -0700 Subject: [PATCH 18/70] feat(desktop): enable llama.cpp user workflows Signed-off-by: Sherief Farouk --- desktop/src/shared/constants/engines.ts | 5 ++-- .../src/ui/assets/engine-icons/llama-cpp.svg | 11 +++++++ desktop/src/ui/components/EngineIcon.tsx | 9 ++++++ desktop/tests/modular/engine-identity.test.ts | 12 ++++---- services/nvpair-tui/proxyengines_test.go | 2 +- services/nvpair-ui-broker/engineproxy_test.go | 6 ++-- services/shared/engines/engines.go | 2 +- services/shared/engines/engines_test.go | 2 +- services/tests/broker_supervision_test.go | 29 ++++++++++--------- services/tests/llamacpp_interop_test.go | 23 +++------------ services/tests/model_routing_interop_test.go | 4 +-- .../tests/workload_identity_interop_test.go | 3 +- 12 files changed, 57 insertions(+), 51 deletions(-) create mode 100644 desktop/src/ui/assets/engine-icons/llama-cpp.svg diff --git a/desktop/src/shared/constants/engines.ts b/desktop/src/shared/constants/engines.ts index 8cf4cad7..15970f2c 100644 --- a/desktop/src/shared/constants/engines.ts +++ b/desktop/src/shared/constants/engines.ts @@ -4,13 +4,12 @@ import { EngineType, ModelExpiry } from '@/shared/types/engines' // The engines `nvpair-engine-manager` ships a manifest for, and therefore the -// only ones PAIR can install, run or route to. Keep staged engines out of -// EnabledEngineTypes until every desktop consumer is ready to expose them. +// only ones PAIR can install, run or route to. export const EngineTypes = ['ollama', 'lm-studio', 'llama-cpp'] as const // Kept as a distinct export so a future engine can ship behind it rather than // appearing the moment its type exists. -export const EnabledEngineTypes: EngineType[] = ['ollama', 'lm-studio'] as const +export const EnabledEngineTypes: EngineType[] = ['ollama', 'lm-studio', 'llama-cpp'] /** * How `nvpair-engine-manager` spells each engine on the wire. This is the diff --git a/desktop/src/ui/assets/engine-icons/llama-cpp.svg b/desktop/src/ui/assets/engine-icons/llama-cpp.svg new file mode 100644 index 00000000..5190d024 --- /dev/null +++ b/desktop/src/ui/assets/engine-icons/llama-cpp.svg @@ -0,0 +1,11 @@ + + + + + + + + + + + diff --git a/desktop/src/ui/components/EngineIcon.tsx b/desktop/src/ui/components/EngineIcon.tsx index 4ee81367..bb1618a0 100644 --- a/desktop/src/ui/components/EngineIcon.tsx +++ b/desktop/src/ui/components/EngineIcon.tsx @@ -5,6 +5,7 @@ import type { CSSProperties } from 'react' import { type EngineType } from '@/shared/types/engines' import ollamaIcon from '@/ui/assets/engine-icons/ollama.png?inline' import lmStudioIcon from '@/ui/assets/engine-icons/lm-studio.png?inline' +import llamaCppIcon from '@/ui/assets/engine-icons/llama-cpp.svg?inline' export default function EngineIcon({ type, size = 32 }: { type: EngineType; size?: number }) { const dimension = `${size}px` @@ -39,5 +40,13 @@ export default function EngineIcon({ type, size = 32 }: { type: EngineType; size ) } + if (type === 'llama-cpp') { + return ( +
+ llama.cpp +
+ ) + } + return null } diff --git a/desktop/tests/modular/engine-identity.test.ts b/desktop/tests/modular/engine-identity.test.ts index 6fd6dec8..bdb6d347 100644 --- a/desktop/tests/modular/engine-identity.test.ts +++ b/desktop/tests/modular/engine-identity.test.ts @@ -12,14 +12,15 @@ import { import { engineManagerName, engineTypeFromManagerName, isEngineType } from '@/shared/utils/engines' import { EngineCapabilities } from '@/ui/constants/engine-capabilities' import { getWelcomeEngineCandidates, WELCOME_ENGINE_DEFAULT_SELECTED } from '@/ui/constants/welcome' +import { gatewayEndpointDisplayUrl } from '@/ui/utils/gateway-inference-paths' describe('engine identity', () => { - it('declares llama.cpp without enabling its desktop workflows', () => { + it('enables llama.cpp workflows without preselecting its large install', () => { expect(EngineTypes).toEqual(['ollama', 'lm-studio', 'llama-cpp']) - expect(EnabledEngineTypes).toEqual(['ollama', 'lm-studio']) - expect(getWelcomeEngineCandidates('Windows')).not.toContain('llama-cpp') - expect(getWelcomeEngineCandidates('MacOS')).not.toContain('llama-cpp') - expect(getWelcomeEngineCandidates('Linux')).not.toContain('llama-cpp') + expect(EnabledEngineTypes).toEqual(['ollama', 'lm-studio', 'llama-cpp']) + expect(getWelcomeEngineCandidates('Windows')).toContain('llama-cpp') + expect(getWelcomeEngineCandidates('MacOS')).toContain('llama-cpp') + expect(getWelcomeEngineCandidates('Linux')).toContain('llama-cpp') expect(WELCOME_ENGINE_DEFAULT_SELECTED['llama-cpp']).toBe(false) }) @@ -53,5 +54,6 @@ describe('engine identity', () => { hasDeleteModel: false, engineHub: { label: 'llama.cpp' } }) + expect(gatewayEndpointDisplayUrl(8080, 'llama-cpp')).toBe('http://127.0.0.1:8080') }) }) diff --git a/services/nvpair-tui/proxyengines_test.go b/services/nvpair-tui/proxyengines_test.go index e36a9897..fcef2300 100644 --- a/services/nvpair-tui/proxyengines_test.go +++ b/services/nvpair-tui/proxyengines_test.go @@ -9,7 +9,7 @@ import ( ) func TestDefaultProxyEngineCSVUsesSharedDefaults(t *testing.T) { - if got, want := defaultProxyEngineCSV(), "ollama,lmstudio"; got != want { + if got, want := defaultProxyEngineCSV(), "ollama,lmstudio,llamacpp"; got != want { t.Fatalf("default proxy engines = %q, want %q", got, want) } } diff --git a/services/nvpair-ui-broker/engineproxy_test.go b/services/nvpair-ui-broker/engineproxy_test.go index 09589c29..a2295740 100644 --- a/services/nvpair-ui-broker/engineproxy_test.go +++ b/services/nvpair-ui-broker/engineproxy_test.go @@ -164,8 +164,8 @@ func TestOwnershipDecidesTheOccupiedFacadeOutcome(t *testing.T) { // --proxy-engines selects which engines the one binary is started for. func TestParseProxyEngines(t *testing.T) { defaults := defaultProxyEngineNames() - if len(defaults) != 2 || defaults[0] != "ollama" || defaults[1] != "lmstudio" { - t.Fatalf("defaultProxyEngineNames() = %v, want [ollama lmstudio]", defaults) + if len(defaults) != 3 || defaults[0] != "ollama" || defaults[1] != "lmstudio" || defaults[2] != "llamacpp" { + t.Fatalf("defaultProxyEngineNames() = %v, want [ollama lmstudio llamacpp]", defaults) } for _, tc := range []struct { name string @@ -173,7 +173,7 @@ func TestParseProxyEngines(t *testing.T) { want []string wantErr bool }{ - {name: "default set", csv: "ollama,lmstudio", want: []string{"ollama", "lmstudio"}}, + {name: "default set", csv: "ollama,lmstudio,llamacpp", want: []string{"ollama", "lmstudio", "llamacpp"}}, {name: "single engine", csv: "lmstudio", want: []string{"lmstudio"}}, {name: "whitespace and blanks are tolerated", csv: " ollama , , lmstudio ", want: []string{"ollama", "lmstudio"}}, {name: "duplicates collapse", csv: "ollama,ollama", want: []string{"ollama"}}, diff --git a/services/shared/engines/engines.go b/services/shared/engines/engines.go index ae2a9872..8be55970 100644 --- a/services/shared/engines/engines.go +++ b/services/shared/engines/engines.go @@ -156,7 +156,7 @@ var all = []Engine{ FacadePort: 8080, EnginePortBase: 8081, PortFile: "llamacpp-proxy-port.json", - ProxyEnabledByDefault: false, + ProxyEnabledByDefault: true, }, } diff --git a/services/shared/engines/engines_test.go b/services/shared/engines/engines_test.go index 5020586e..5b6abb39 100644 --- a/services/shared/engines/engines_test.go +++ b/services/shared/engines/engines_test.go @@ -36,7 +36,7 @@ func TestNames(t *testing.T) { func TestProxyDefaults(t *testing.T) { got := ProxyDefaults() - want := []string{"ollama", "lmstudio"} + want := []string{"ollama", "lmstudio", "llamacpp"} if len(got) != len(want) { t.Fatalf("ProxyDefaults() = %v, want %v", got, want) } diff --git a/services/tests/broker_supervision_test.go b/services/tests/broker_supervision_test.go index 87a9fce0..98613274 100644 --- a/services/tests/broker_supervision_test.go +++ b/services/tests/broker_supervision_test.go @@ -276,22 +276,22 @@ func mustPid(t *testing.T, line string) int { return pid } -// Both engines are fronted by ONE process. This is the property the whole +// Every default engine is fronted by ONE process. This is the property the whole // unification exists for: the estimated-work reservations a proxy takes -// between scheduler snapshots live in the process, so two processes each held -// half the picture and could dispatch simultaneous bursts to the same node -// each believing it idle. +// between scheduler snapshots live in the process, so separate processes each +// held an incomplete picture and could dispatch simultaneous bursts to the +// same node while each believed it idle. // // Asserted on pids rather than on the code shape, because "one supervisor" is -// an implementation detail and "one OS process serving both engines" is the +// an implementation detail and "one OS process serving every engine" is the // thing that actually makes the reservation map whole. -func TestBothEnginesAreServedByOneProxyProcess(t *testing.T) { +func TestAllDefaultEnginesAreServedByOneProxyProcess(t *testing.T) { if portBusy(11435) || portBusy(1234) { t.Skip("ollama-proxy (11435) or lmstudio-proxy (1234) default port already in use; skipping") } stdin, msgs, stderr, cleanup := startBrokerWith(t, - // No --proxy-engines: both engines, which is the default. + // No --proxy-engines: exercise the complete default set. "--proxy-path", proxyBin, ) t.Cleanup(cleanup) @@ -309,14 +309,17 @@ func TestBothEnginesAreServedByOneProxyProcess(t *testing.T) { waitForMethod(t, msgs, "app:ready", 10*time.Second) - // Both facades must actually come up, or "one process" would be trivially + // Every facade must actually come up, or "one process" would be trivially // true by one of them having failed. ollamaPort := waitProxyReady(t, stdin, msgs, 15*time.Second) lmstudioPort := waitLMStudioProxyReady(t, stdin, msgs, 15*time.Second) - if ollamaPort == lmstudioPort { - t.Fatalf("both facades report port %d; they must bind separately", ollamaPort) + llamacppPort := waitEngineProxyReady(t, "llamacpp-proxy", stdin, msgs, 15*time.Second) + if len(map[int]bool{ollamaPort: true, lmstudioPort: true, llamacppPort: true}) != 3 { + t.Fatalf("facade ports must be distinct: ollama=%d lmstudio=%d llamacpp=%d", + ollamaPort, lmstudioPort, llamacppPort) } - t.Logf("ollama facade on :%d, lmstudio facade on :%d", ollamaPort, lmstudioPort) + t.Logf("facades ready: ollama=%d lmstudio=%d llamacpp=%d", + ollamaPort, lmstudioPort, llamacppPort) // Collect every spawn announced up to this point plus a settling window, so // a second process starting late still shows up. @@ -331,9 +334,9 @@ func TestBothEnginesAreServedByOneProxyProcess(t *testing.T) { } } if len(seen) != 1 { - t.Fatalf("saw %d proxy processes %v, want exactly 1 hosting both engines", len(seen), seen) + t.Fatalf("saw %d proxy processes %v, want exactly 1 hosting all default engines", len(seen), seen) } - t.Logf("both engines served by pid %v", seen) + t.Logf("all default engines served by pid %v", seen) } // TestBrokerShutsDownOnSignal verifies the broker exits promptly when it diff --git a/services/tests/llamacpp_interop_test.go b/services/tests/llamacpp_interop_test.go index 4c3c5c58..86f988dc 100644 --- a/services/tests/llamacpp_interop_test.go +++ b/services/tests/llamacpp_interop_test.go @@ -15,7 +15,7 @@ import ( "time" ) -func TestLlamaCPPProxyIsExcludedFromBrokerDefaults(t *testing.T) { +func TestLlamaCPPProxyIsIncludedInBrokerDefaults(t *testing.T) { stdin, msgs, stderr, cleanup := startBrokerWith(t, "--proxy-path", proxyBin) t.Cleanup(cleanup) go func() { @@ -24,27 +24,12 @@ func TestLlamaCPPProxyIsExcludedFromBrokerDefaults(t *testing.T) { }() waitForMethod(t, msgs, "app:ready", 10*time.Second) - const requestID = 7100 - writeRawFrame(t, stdin, fmt.Sprintf( - `{"jsonrpc":"2.0","id":%d,"method":"llamacpp-proxy:get-status"}`, requestID, - )) - response := waitForResponseID(t, msgs, requestID, 5*time.Second) - if response.Error != nil { - t.Fatalf("llamacpp-proxy:get-status failed: %d %s", response.Error.Code, response.Error.Message) - } - var status struct { - Ready bool `json:"ready"` - Port int `json:"port"` - } - if err := json.Unmarshal(response.Result, &status); err != nil { - t.Fatalf("decode llama.cpp status %s: %v", response.Result, err) - } - if status.Ready || status.Port != 0 { - t.Fatalf("default llama.cpp status = %+v, want disabled", status) + if port := waitEngineProxyReady(t, "llamacpp-proxy", stdin, msgs, 15*time.Second); port <= 0 { + t.Fatalf("default llama.cpp proxy port = %d, want a listening facade", port) } } -func TestLlamaCPPOptInFacadeUsesRouterInventoryAndExactModelIDs(t *testing.T) { +func TestLlamaCPPFacadeUsesRouterInventoryAndExactModelIDs(t *testing.T) { const model = "org/router-model-GGUF:Q4_K_M" var modelListHits atomic.Int32 var inferenceHits atomic.Int32 diff --git a/services/tests/model_routing_interop_test.go b/services/tests/model_routing_interop_test.go index 58ef22ea..58056042 100644 --- a/services/tests/model_routing_interop_test.go +++ b/services/tests/model_routing_interop_test.go @@ -48,9 +48,7 @@ func TestStrictModelRoutingAcrossProcesses(t *testing.T) { ineligible := newRoutingUpstream(t, http.StatusOK) stdin, msgs, stderr, cleanup := startBrokerWith(t, - // No --proxy-engines: this test wants both engines fronted, which is - // the default. - "--proxy-path", proxyBin, + "--proxy-path", proxyBin, "--proxy-engines", "ollama,lmstudio", ) t.Cleanup(cleanup) go func() { diff --git a/services/tests/workload_identity_interop_test.go b/services/tests/workload_identity_interop_test.go index 864a3e58..2300c356 100644 --- a/services/tests/workload_identity_interop_test.go +++ b/services/tests/workload_identity_interop_test.go @@ -66,8 +66,7 @@ func TestWorkloadCrossEngineIdentityDistinct(t *testing.T) { lmstudioPort := portOfURL(t, lmstudio.URL) stdin, msgs, _, cleanup := startBrokerWith(t, - // Both engines fronted, which is the default. - "--proxy-path", proxyBin, + "--proxy-path", proxyBin, "--proxy-engines", "ollama,lmstudio", "--workload-manager-path", workloadMgrBin, ) t.Cleanup(cleanup) From 93b93dbfeee4c709f051a99d25f885bb96d09674 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 20:34:16 -0700 Subject: [PATCH 19/70] feat(demo): route inference demo through llama.cpp Signed-off-by: Sherief Farouk --- desktop/src/electron/inference-demo.ts | 8 +-- desktop/src/shared/types/inference-demo.ts | 10 +-- .../src/shared/types/inference-dispatcher.ts | 2 +- .../modular/inference-demo-lifecycle.test.ts | 38 ++++++++-- .../modular/inference-demo-schedule.test.ts | 16 +++-- scripts/inference-dispatcher/client.go | 35 +++++++--- scripts/inference-dispatcher/config.go | 16 +++-- .../inference-dispatcher/dispatcher_test.go | 69 +++++++++++++++++++ 8 files changed, 159 insertions(+), 35 deletions(-) diff --git a/desktop/src/electron/inference-demo.ts b/desktop/src/electron/inference-demo.ts index 37e39408..cb9c4b49 100644 --- a/desktop/src/electron/inference-demo.ts +++ b/desktop/src/electron/inference-demo.ts @@ -28,9 +28,9 @@ import { buildDemoSchedule, type ScheduledRequest } from '@/electron/inference-d * Inference Demo scheduler (node-local). * * Owns the open-loop submission schedule described in the mini spec. The Go - * `inference-dispatcher` binary is unchanged: it still runs one backend/model - * per invocation, so this module spawns one short-lived dispatcher process per - * scheduled request and lets NV PAIR do all queueing and routing. + * `inference-dispatcher` runs one engine/model per invocation, so this module + * spawns one short-lived dispatcher process per scheduled request and lets + * NV PAIR do all queueing and routing. * * Three invariants matter more than anything else here: * @@ -366,7 +366,7 @@ export async function startInferenceDemo(): Promise { // Deliberately names no port: the proxies own their listeners, and // quoting a number here would be the same mistake as hardcoding one. throw new Error( - 'No local inference engine exposed a text-generation model. Start Ollama or LM Studio, wait for it to appear in Settings, and try again.' + 'No local inference engine exposed a text-generation model. Start Ollama, LM Studio, or llama.cpp, wait for it to appear in Settings, and try again.' ) } diff --git a/desktop/src/shared/types/inference-demo.ts b/desktop/src/shared/types/inference-demo.ts index 4af2f35a..6a13373e 100644 --- a/desktop/src/shared/types/inference-demo.ts +++ b/desktop/src/shared/types/inference-demo.ts @@ -2,6 +2,7 @@ // SPDX-License-Identifier: Apache-2.0 import type { DispatcherBackend } from '@/shared/types/inference-dispatcher' +import type { EngineType } from '@/shared/types/engines' /** * Inference Demo — a fixed, node-local burst of synthetic inference traffic sent @@ -17,8 +18,8 @@ import type { DispatcherBackend } from '@/shared/types/inference-dispatcher' * status toast. * * Scheduling lives entirely in the Electron main process - * (`@/electron/inference-demo`). The Go `inference-dispatcher` binary is - * unchanged and is invoked once per scheduled request group. + * (`@/electron/inference-demo`). The Go `inference-dispatcher` binary is invoked + * once per scheduled request. */ /** Wall-clock offsets, in seconds, at which a cohort of simulated agents starts. */ @@ -54,10 +55,11 @@ export const DEMO_REQUEST_TIMEOUT_SECONDS = 120 export const DEMO_ENGINE_PROBES: readonly { backend: DispatcherBackend /** Engine key used by the broker's proxy port registry. */ - proxyEngine: 'ollama' | 'lm-studio' + proxyEngine: EngineType }[] = [ { backend: 'ollama', proxyEngine: 'ollama' }, - { backend: 'lmstudio', proxyEngine: 'lm-studio' } + { backend: 'lmstudio', proxyEngine: 'lm-studio' }, + { backend: 'llamacpp', proxyEngine: 'llama-cpp' } ] /** diff --git a/desktop/src/shared/types/inference-dispatcher.ts b/desktop/src/shared/types/inference-dispatcher.ts index 2abe35ae..9c6d7a45 100644 --- a/desktop/src/shared/types/inference-dispatcher.ts +++ b/desktop/src/shared/types/inference-dispatcher.ts @@ -10,7 +10,7 @@ * reads its `--list-models` inventory. */ -export type DispatcherBackend = 'ollama' | 'lmstudio' +export type DispatcherBackend = 'ollama' | 'lmstudio' | 'llamacpp' /** One entry from the binary's `--list-models` JSON inventory. */ export interface DispatcherModel { diff --git a/desktop/tests/modular/inference-demo-lifecycle.test.ts b/desktop/tests/modular/inference-demo-lifecycle.test.ts index d9c4b95e..54cb20fb 100644 --- a/desktop/tests/modular/inference-demo-lifecycle.test.ts +++ b/desktop/tests/modular/inference-demo-lifecycle.test.ts @@ -2,6 +2,7 @@ // SPDX-License-Identifier: Apache-2.0 import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { EngineType } from '@/shared/types/engines' /** * Lifecycle guarantees for the Inference Demo scheduler. @@ -30,6 +31,7 @@ const spawned: SpawnRecord[] = [] /** Options each `--list-models` discovery probe was invoked with. */ const probeOptions: ChildOptions[] = [] +const probeArgs: string[][] = [] /** * Poisoned parent environment. `INFERENCE_DISPATCHER_LOOP` would run the child @@ -45,9 +47,10 @@ const POISONED_ENV: Record = { } /** Proxy ports the fake broker reports. Mutable so a test can withhold one. */ -const proxyPorts: Record<'ollama' | 'lm-studio', number | null> = { +const proxyPorts: Record = { ollama: 11434, - 'lm-studio': 1234 + 'lm-studio': 1234, + 'llama-cpp': 8080 } /** Model inventory each probe returns. Mutable so a test can return none. */ @@ -96,6 +99,7 @@ vi.mock('node:child_process', () => ({ options: ChildOptions, callback: (error: Error | null, stdout: string, stderr: string) => void ) => { + probeArgs.push(_args) probeOptions.push(options) callback(null, JSON.stringify(inventory), '') return { exitCode: null, once: () => {}, kill: () => true } @@ -114,7 +118,7 @@ vi.mock('electron', () => ({ vi.mock('@/electron/service-bridge/modular-state', () => ({ getModularBridgeState: () => ({ - getProxyPort: (engine: 'ollama' | 'lm-studio') => proxyPorts[engine] + getProxyPort: (engine: EngineType) => proxyPorts[engine] }) })) @@ -126,18 +130,24 @@ import { } from '@/electron/inference-demo' /** Ports the demo is allowed to target: proxy facades only. */ -const PROXY_FACADE_PORTS = [11434, 1234] +const PROXY_FACADE_PORTS = [11434, 1234, 8080] function portOf(record: SpawnRecord): number { return Number(record.args[record.args.indexOf('--port') + 1]) } +function backendOf(record: SpawnRecord): string { + return record.args[record.args.indexOf('--backend') + 1] ?? '' +} + beforeEach(() => { spawned.length = 0 probeOptions.length = 0 + probeArgs.length = 0 inventory = [{ name: 'demo-model', type: 'llm' }] proxyPorts.ollama = 11434 proxyPorts['lm-studio'] = 1234 + proxyPorts['llama-cpp'] = 8080 Object.assign(process.env, POISONED_ENV) vi.useFakeTimers() }) @@ -203,14 +213,30 @@ describe('inference demo lifecycle', () => { } }) + it('probes and schedules llama.cpp through its proxy facade', async () => { + await startInferenceDemo() + await vi.advanceTimersByTimeAsync(12_000) + + expect( + probeArgs.some( + args => + args[args.indexOf('--backend') + 1] === 'llamacpp' && + args.includes('--list-models') + ) + ).toBe(true) + const llamaCPPRequest = spawned.find(child => backendOf(child) === 'llamacpp') + expect(llamaCPPRequest).toBeDefined() + if (llamaCPPRequest) expect(portOf(llamaCPPRequest)).toBe(8080) + }) + it('skips an engine whose proxy has not reported a port', async () => { - proxyPorts['lm-studio'] = null + proxyPorts['llama-cpp'] = null await startInferenceDemo() await vi.advanceTimersByTimeAsync(70_000) expect(spawned.length).toBeGreaterThan(0) for (const child of spawned) { - expect(portOf(child)).toBe(11434) + expect(portOf(child)).not.toBe(8080) } }) diff --git a/desktop/tests/modular/inference-demo-schedule.test.ts b/desktop/tests/modular/inference-demo-schedule.test.ts index 359dcdba..6d93a6be 100644 --- a/desktop/tests/modular/inference-demo-schedule.test.ts +++ b/desktop/tests/modular/inference-demo-schedule.test.ts @@ -13,11 +13,17 @@ import { // Ports are the proxy facades PAIR routes through, never an engine's own // backend port. At runtime these come from the broker's proxy registry. function targets(count: number): DemoTarget[] { - return Array.from({ length: count }, (_, index) => ({ - backend: index % 2 === 0 ? ('ollama' as const) : ('lmstudio' as const), - port: index % 2 === 0 ? 11434 : 1234, - model: `model-${index}` - })) + return Array.from({ length: count }, (_, index) => { + const model = `model-${index}` + switch (index % 3) { + case 0: + return { backend: 'ollama', port: 11434, model } + case 1: + return { backend: 'lmstudio', port: 1234, model } + default: + return { backend: 'llamacpp', port: 8080, model } + } + }) } function key(target: DemoTarget): string { diff --git a/scripts/inference-dispatcher/client.go b/scripts/inference-dispatcher/client.go index a759e973..f7eae1fd 100644 --- a/scripts/inference-dispatcher/client.go +++ b/scripts/inference-dispatcher/client.go @@ -111,9 +111,12 @@ func decodeObject(data []byte, target any) error { func (c *backendClient) listModels(ctx context.Context) ([]RegisteredModel, error) { var models []RegisteredModel var err error - if c.cfg.Backend == "lmstudio" { + switch c.cfg.Backend { + case "lmstudio": models, err = c.listLMStudioModels(ctx) - } else { + case "llamacpp": + models, err = c.listLlamaCPPModels(ctx) + default: models, err = c.listOllamaModels(ctx) } if err != nil { @@ -128,6 +131,14 @@ func (c *backendClient) listModels(ctx context.Context) ([]RegisteredModel, erro return models, nil } +func (c *backendClient) listLlamaCPPModels(ctx context.Context) ([]RegisteredModel, error) { + data, err := c.request(ctx, http.MethodGet, "/v1/models", nil) + if err != nil { + return nil, fmt.Errorf("query llama.cpp models: %w", err) + } + return parseOpenAIModels(data) +} + func (c *backendClient) listOllamaModels(ctx context.Context) ([]RegisteredModel, error) { data, err := c.request(ctx, http.MethodGet, "/api/tags", nil) if err != nil { @@ -173,7 +184,7 @@ func (c *backendClient) listLMStudioModels(ctx context.Context) ([]RegisteredMod openAIData, openAIErr := c.request(ctx, http.MethodGet, "/v1/models", nil) var models []RegisteredModel if openAIErr == nil { - parsed, err := parseLMStudioOpenAIModels(openAIData) + parsed, err := parseOpenAIModels(openAIData) if err != nil { openAIErr = err } else if len(parsed) == 0 { @@ -251,7 +262,7 @@ func parseLMStudioNativeModels(data []byte) ([]RegisteredModel, error) { return models, nil } -func parseLMStudioOpenAIModels(data []byte) ([]RegisteredModel, error) { +func parseOpenAIModels(data []byte) ([]RegisteredModel, error) { var response struct { Data []struct { ID string `json:"id"` @@ -341,15 +352,19 @@ func (c *backendClient) resolveModel(ctx context.Context) (string, []RegisteredM } func (c *backendClient) inferencePath() string { - if c.cfg.Backend == "lmstudio" { + if c.usesOpenAIProtocol() { return "/v1/chat/completions" } return "/api/generate" } +func (c *backendClient) usesOpenAIProtocol() bool { + return c.cfg.Backend == "lmstudio" || c.cfg.Backend == "llamacpp" +} + func (c *backendClient) infer(ctx context.Context, model, prompt string) (string, error) { var payload map[string]any - if c.cfg.Backend == "lmstudio" { + if c.usesOpenAIProtocol() { payload = map[string]any{ "model": model, "messages": []map[string]string{{"role": "user", "content": prompt}}, @@ -386,8 +401,8 @@ func (c *backendClient) infer(ctx context.Context, model, prompt string) (string if err != nil { return "", err } - if c.cfg.Backend == "lmstudio" { - return parseLMStudioResponse(data) + if c.usesOpenAIProtocol() { + return parseOpenAIResponse(data) } var response struct { Response string `json:"response"` @@ -398,7 +413,7 @@ func (c *backendClient) infer(ctx context.Context, model, prompt string) (string return strings.TrimSpace(response.Response), nil } -func parseLMStudioResponse(data []byte) (string, error) { +func parseOpenAIResponse(data []byte) (string, error) { var response struct { Choices []struct { Message struct { @@ -411,7 +426,7 @@ func parseLMStudioResponse(data []byte) (string, error) { return "", err } if len(response.Choices) == 0 { - return "", errors.New("LM Studio response contained no choices") + return "", errors.New("OpenAI-compatible response contained no choices") } choice := response.Choices[0] switch content := choice.Message.Content.(type) { diff --git a/scripts/inference-dispatcher/config.go b/scripts/inference-dispatcher/config.go index 2598dcbc..892f313c 100644 --- a/scripts/inference-dispatcher/config.go +++ b/scripts/inference-dispatcher/config.go @@ -23,6 +23,8 @@ const ( // PAIR's managed LM Studio backend is moved behind it starting at 1235; // pass --port explicitly to reach that directly, which bypasses routing. defaultLMStudioPort = 1234 + // llama.cpp's OpenAI-compatible proxy facade uses its stock router port. + defaultLlamaCPPPort = 8080 defaultErrorLog = "inference_errors.txt" maxPromptsPerBatch = 100 ) @@ -289,7 +291,7 @@ func parseConfig(args []string, stderr io.Writer) (Config, error) { fs.SetOutput(stderr) var parsedConfigPath string fs.StringVar(&parsedConfigPath, "config", configPath, "JSON configuration file") - fs.StringVar(&cfg.Backend, "backend", cfg.Backend, "backend: ollama or lmstudio") + fs.StringVar(&cfg.Backend, "backend", cfg.Backend, "backend: ollama, lmstudio, or llamacpp") fs.StringVar(&cfg.Backend, "provider", cfg.Backend, "alias for --backend") fs.IntVar(&cfg.Port, "port", cfg.Port, "server port (backend default when omitted)") fs.StringVar(&cfg.Model, "model", cfg.Model, "model name; omitted or auto selects an available model") @@ -356,8 +358,8 @@ func parseConfig(args []string, stderr io.Writer) (Config, error) { } func validateConfig(cfg Config) error { - if cfg.Backend != "ollama" && cfg.Backend != "lmstudio" { - return errors.New("--backend must be ollama or lmstudio") + if cfg.Backend != "ollama" && cfg.Backend != "lmstudio" && cfg.Backend != "llamacpp" { + return errors.New("--backend must be ollama, lmstudio, or llamacpp") } if cfg.Port < 0 || cfg.Port > 65535 { return errors.New("--port must be between 1 and 65535") @@ -414,8 +416,12 @@ func effectivePort(cfg Config) int { if cfg.Port != 0 { return cfg.Port } - if cfg.Backend == "lmstudio" { + switch cfg.Backend { + case "lmstudio": return defaultLMStudioPort + case "llamacpp": + return defaultLlamaCPPPort + default: + return defaultOllamaPort } - return defaultOllamaPort } diff --git a/scripts/inference-dispatcher/dispatcher_test.go b/scripts/inference-dispatcher/dispatcher_test.go index c858fcc7..60ae4124 100644 --- a/scripts/inference-dispatcher/dispatcher_test.go +++ b/scripts/inference-dispatcher/dispatcher_test.go @@ -152,6 +152,64 @@ func TestLMStudioFallsBackToOpenAIInventory(t *testing.T) { } } +func TestLlamaCPPUsesOpenAIInventoryAndChat(t *testing.T) { + type observedRequest struct { + method string + path string + model string + messageCount int + } + observed := make(chan observedRequest, 2) + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + switch { + case r.Method == http.MethodGet && r.URL.Path == "/v1/models": + observed <- observedRequest{method: r.Method, path: r.URL.Path} + _, _ = w.Write([]byte(`{"data":[{"id":"llama-demo"}]}`)) + case r.Method == http.MethodPost && r.URL.Path == "/v1/chat/completions": + var request struct { + Model string `json:"model"` + Messages []struct { + Role string `json:"role"` + Content string `json:"content"` + } `json:"messages"` + } + if err := json.NewDecoder(r.Body).Decode(&request); err != nil { + t.Errorf("decode chat request: %v", err) + } + observed <- observedRequest{ + method: r.Method, + path: r.URL.Path, + model: request.Model, + messageCount: len(request.Messages), + } + _, _ = w.Write([]byte(`{"choices":[{"message":{"content":"done"}}]}`)) + default: + http.NotFound(w, r) + } + })) + defer server.Close() + + var stdout, stderr bytes.Buffer + exit := runAgainstServer( + t, + context.Background(), + []string{"--backend", "llamacpp", "--prompt", "test"}, + server.URL, + &stdout, + &stderr, + ) + if exit != 0 { + t.Fatalf("exit=%d stderr=%s", exit, stderr.String()) + } + if got := <-observed; got.method != http.MethodGet || got.path != "/v1/models" { + t.Fatalf("inventory request = %+v, want GET /v1/models", got) + } + if got := <-observed; got.method != http.MethodPost || got.path != "/v1/chat/completions" || + got.model != "llama-demo" || got.messageCount != 1 { + t.Fatalf("inference request = %+v, want OpenAI chat for llama-demo", got) + } +} + // A Personal AI Router proxy answers /v1/models with the whole cluster's // inventory and forwards /api/v1/models to one node, so the aggregated list must // win. The native list still supplies the type and capability fields the @@ -334,6 +392,17 @@ func TestLMStudioDefaultPort(t *testing.T) { } } +func TestLlamaCPPDefaultPort(t *testing.T) { + var stderr bytes.Buffer + cfg, err := parseConfig([]string{"--backend", "llamacpp"}, &stderr) + if err != nil { + t.Fatalf("parse config: %v", err) + } + if port := effectivePort(cfg); port != 8080 { + t.Fatalf("llama.cpp default port=%d, want 8080", port) + } +} + func TestResponseTextNeverReachesStdout(t *testing.T) { const secret = "the capital of France is Paris" server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { From 207e0494606a0f9e05a65fe93bed99de151480b1 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Thu, 24 Sep 2026 22:01:03 -0700 Subject: [PATCH 20/70] docs: publish llama.cpp application support Signed-off-by: Sherief Farouk --- .cursor/rules/model-registry.mdc | 16 +++++--- .cursor/rules/system-architecture.mdc | 25 +++++++----- README.md | 23 ++++++----- desktop/docs/architecture.md | 11 ++++-- desktop/docs/frontend-api.md | 11 +++++- desktop/docs/services-backend.md | 11 +++--- desktop/docs/services-parity.md | 38 ++++++++++++------- docs/architecture.mdx | 26 +++++++------ docs/engine-lifecycle.mdx | 27 ++++++++----- docs/engine-settings.mdx | 8 ++-- docs/getting-started.mdx | 33 ++++++++++------ docs/inference-dispatcher.mdx | 16 +++++--- docs/overview.mdx | 8 ++-- docs/terminal-interface.mdx | 15 ++++---- services/nvpair-engine-manager/LAUNCH_TEXT.md | 10 +++-- services/nvpair-job-scheduler/README.md | 9 +++-- services/nvpair-job-scheduler/spec.md | 9 +++-- services/nvpair-proxy/README.md | 10 ++--- services/nvpair-proxy/spec.md | 2 +- services/nvpair-tui/README.md | 6 +-- services/nvpair-ui-broker/ENGINE_SETTINGS.md | 16 ++++---- services/nvpair-ui-broker/README.md | 6 +-- services/nvpair-workload-manager/spec.md | 2 +- services/readme.md | 11 +++--- 24 files changed, 211 insertions(+), 138 deletions(-) diff --git a/.cursor/rules/model-registry.mdc b/.cursor/rules/model-registry.mdc index e66521a7..9bddc017 100644 --- a/.cursor/rules/model-registry.mdc +++ b/.cursor/rules/model-registry.mdc @@ -1,5 +1,5 @@ --- -description: Electron-main engine model hub (Ollama committed list + LM Studio catalog) +description: Electron-main engine model hub (locked Ollama/llama.cpp + live LM Studio) alwaysApply: true --- (HTTPS download + verify-if-pinned + user-mode run) --> Stopped Stopped --engine:start----> (adopt if already serving the port, else spawn) --> Running --health--> Running -Running --engine:stop-----> (stop signal, wait for exit; no timeout) --> Stopped +Running --engine:stop-----> (stop signal, bounded grace, force if needed) --> Stopped ``` Detect uses the manifest's `detect` paths. Install is one-shot and @@ -157,11 +157,12 @@ unexpected exit is reported. The bundled Ollama manifest allows up to ten minutes for startup because GPU discovery can exceed the previous 30-second allowance on supported Windows systems. The deadline remains finite: if Ollama never serves its readiness endpoint, engine-manager stops the owned process and -reports the failed start. Stop sends one stop signal and waits for the engine -to exit, with no timeout: SIGTERM to the process group on Unix (graceful, no -SIGKILL escalation), and `taskkill /T /F` on Windows — where the windowless -engines we spawn can't receive a graceful (non-`/F`) close, so a forced -terminate is the only signal that actually stops them. +reports the failed start. On Unix, stopping an owned process sends SIGTERM to +its process group, waits the manifest's `stop.grace_s` (five seconds by +default), then escalates to SIGKILL; `signal:"kill"` skips the grace. Failed +startup cleanup uses the same policy. On Windows, the windowless engines we +spawn cannot receive a graceful (non-`/F`) close, so stopping uses immediate +`taskkill /T /F`. ### Adoption — start may attach to an engine it didn't launch diff --git a/services/nvpair-engine-manager/spec.md b/services/nvpair-engine-manager/spec.md index 28ad6c58..83e1e6b8 100644 --- a/services/nvpair-engine-manager/spec.md +++ b/services/nvpair-engine-manager/spec.md @@ -82,6 +82,7 @@ extensibility story for an open-source product. **Functional** - Load + validate per-engine JSON manifests (bundled + user dir); select the host `/` block; resolve placeholders (`{bin}`, `{cli}`, `{port}`, `{download}`, `{install_dir}`). Install commands also receive resolved download and destination paths in child-scoped `NVPAIR_INSTALL_*` environment variables so shell reparsing cannot corrupt them. - Support both `process` (owned foreground) and `command` (daemon + control-CLI) runtimes; execute detect / install / uninstall / start / stop / restart / status / health and HTTP **or** CLI actions; emit `engine:*` results and notifications. +- Bound process-mode stops: on Unix send SIGTERM to the owned process group, then SIGKILL after `runtime.stop.grace_s` (five seconds by default); `signal:"kill"` skips the grace. On Windows, windowless managed engines require immediate `taskkill /T /F`. - Emit `errors:report` / `errors:clear` on its stdio for the Broker to forward to `nvpair-errors`. **Non-functional** From b1389217229d18a9ea46a5c6e58ddd0fd4e4b6de Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Fri, 2 Oct 2026 13:46:00 -0700 Subject: [PATCH 52/70] fix(engine-manager): scope artifact placeholders by platform Signed-off-by: Sherief Farouk --- services/nvpair-engine-manager/registry.go | 91 ++++++++++--------- .../nvpair-engine-manager/registry_test.go | 27 ++++++ 2 files changed, 74 insertions(+), 44 deletions(-) diff --git a/services/nvpair-engine-manager/registry.go b/services/nvpair-engine-manager/registry.go index 51745e88..c1025385 100644 --- a/services/nvpair-engine-manager/registry.go +++ b/services/nvpair-engine-manager/registry.go @@ -818,23 +818,29 @@ func (a *Action) validate(name string) error { return nil } -// validatePlaceholders rejects any `{token}` outside allowedPlaceholders -// across every templated string in the manifest. +// validatePlaceholders rejects any `{token}` outside the placeholders +// available to the platform or manifest-global action that contains it. func (m *Manifest) validatePlaceholders() error { - allowed := make(map[string]bool, len(allowedPlaceholders)) - for name := range allowedPlaceholders { - allowed[name] = true - } - for _, platform := range m.Platforms { - if platform.Install == nil { - continue + for key, platform := range m.Platforms { + allowed := make(map[string]bool, len(allowedPlaceholders)) + for name := range allowedPlaceholders { + allowed[name] = true + } + if platform.Install != nil { + for _, artifact := range platform.Install.Artifacts { + allowed["download_"+artifact.Name] = true + } } - for _, artifact := range platform.Install.Artifacts { - allowed["download_"+artifact.Name] = true + if err := validatePlaceholderStrings(platform.templatedStrings(), allowed); err != nil { + return fmt.Errorf("platform %q: %w", key, err) } } - for _, s := range m.templatedStrings() { - for _, match := range placeholderRe.FindAllStringSubmatch(s, -1) { + return validatePlaceholderStrings(m.actionTemplatedStrings(), allowedPlaceholders) +} + +func validatePlaceholderStrings(templates []string, allowed map[string]bool) error { + for _, template := range templates { + for _, match := range placeholderRe.FindAllStringSubmatch(template, -1) { if !allowed[match[1]] { return fmt.Errorf("unknown placeholder {%s} (allowed: %s)", match[1], strings.Join(placeholderList(allowed), ", ")) } @@ -843,12 +849,6 @@ func (m *Manifest) validatePlaceholders() error { return nil } -// allowedPlaceholderList returns the allowed placeholder names, sorted, -// so error messages can't drift from the actual allow-set. -func allowedPlaceholderList() []string { - return placeholderList(allowedPlaceholders) -} - func placeholderList(placeholders map[string]bool) []string { out := make([]string, 0, len(placeholders)) for k := range placeholders { @@ -858,33 +858,36 @@ func placeholderList(placeholders map[string]bool) []string { return out } -// templatedStrings collects every string the runner resolves -// placeholders in, so validatePlaceholders can scan them all. -func (m *Manifest) templatedStrings() []string { +// templatedStrings collects every platform-local string the runner resolves +// placeholders in, so artifact placeholders stay scoped to their platform. +func (p Platform) templatedStrings() []string { var out []string - for _, p := range m.Platforms { - out = append(out, p.Detect...) - if p.Install != nil { - out = append(out, p.Install.Run...) - out = append(out, p.Install.Script...) - } - if p.Uninstall != nil { - out = append(out, p.Uninstall.Run...) - } - out = append(out, p.Runtime.Bin) - out = append(out, p.Runtime.Args...) - for _, cmd := range p.Runtime.Start { - out = append(out, cmd...) - } - for _, v := range p.Runtime.Env { - out = append(out, v) - } - out = append(out, probeStrings(p.Runtime.Ready)...) - out = append(out, probeStrings(p.Runtime.Health)...) - if p.Runtime.Stop != nil { - out = append(out, p.Runtime.Stop.Cmd...) - } + out = append(out, p.Detect...) + if p.Install != nil { + out = append(out, p.Install.Run...) + out = append(out, p.Install.Script...) + } + if p.Uninstall != nil { + out = append(out, p.Uninstall.Run...) + } + out = append(out, p.Runtime.Bin) + out = append(out, p.Runtime.Args...) + for _, cmd := range p.Runtime.Start { + out = append(out, cmd...) + } + for _, v := range p.Runtime.Env { + out = append(out, v) } + out = append(out, probeStrings(p.Runtime.Ready)...) + out = append(out, probeStrings(p.Runtime.Health)...) + if p.Runtime.Stop != nil { + out = append(out, p.Runtime.Stop.Cmd...) + } + return out +} + +func (m *Manifest) actionTemplatedStrings() []string { + var out []string for _, act := range m.Actions { if act.RemovePath != nil { out = append(out, act.RemovePath.Root) diff --git a/services/nvpair-engine-manager/registry_test.go b/services/nvpair-engine-manager/registry_test.go index b3611c53..eccf40d2 100644 --- a/services/nvpair-engine-manager/registry_test.go +++ b/services/nvpair-engine-manager/registry_test.go @@ -156,6 +156,33 @@ func TestValidateAcceptsNamedInstallArtifacts(t *testing.T) { } } +func TestValidateRejectsArtifactPlaceholderFromAnotherPlatform(t *testing.T) { + m := validManifest() + setInstallArtifacts(&m, validInstallArtifacts()) + mac := Platform{ + Install: &Install{ + Fetch: &Fetch{URL: "https://example/server.tar.gz"}, + Run: []string{"extract", "{download}"}, + }, + Runtime: Runtime{Bin: "{install_dir}/llama-server"}, + } + m.Platforms["darwin/arm64"] = mac + if err := m.Validate(); err != nil { + t.Fatalf("valid multi-platform fixture rejected: %v", err) + } + + mac.Install.Run = []string{"extract", "{download_cudart}"} + m.Platforms["darwin/arm64"] = mac + err := m.Validate() + if err == nil { + t.Fatal("macOS install references {download_cudart}, but only Linux declares cudart; want validation error") + } + const want = `platform "darwin/arm64": unknown placeholder {download_cudart}` + if !strings.Contains(err.Error(), want) { + t.Fatalf("error = %q, want it to contain %q", err, want) + } +} + func validInstallArtifacts() []InstallArtifact { return []InstallArtifact{ {Name: "server", URL: "https://example/server.zip", SHA256: strings.Repeat("a", 64)}, From ab42114218b5258af2fc70c164b73e26026fec0f Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Fri, 2 Oct 2026 18:34:06 -0700 Subject: [PATCH 53/70] test(engine-manager): guard artifact install completion events Signed-off-by: Sherief Farouk --- .../install_artifacts_test.go | 52 +++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/services/nvpair-engine-manager/install_artifacts_test.go b/services/nvpair-engine-manager/install_artifacts_test.go index 39e577b5..6f362906 100644 --- a/services/nvpair-engine-manager/install_artifacts_test.go +++ b/services/nvpair-engine-manager/install_artifacts_test.go @@ -14,10 +14,53 @@ import ( "os" "path/filepath" "strings" + "sync" "testing" "time" ) +type installProgressRecorder struct { + mu sync.Mutex + events []any +} + +func (r *installProgressRecorder) emit(method string, params any) { + if method != "engine:install-progress" { + return + } + r.mu.Lock() + r.events = append(r.events, params) + r.mu.Unlock() +} + +func (r *installProgressRecorder) assertSingleTerminal(t *testing.T, want string) { + t.Helper() + r.mu.Lock() + events := append([]any(nil), r.events...) + r.mu.Unlock() + + stages := make([]string, 0, len(events)) + terminal := make([]string, 0, 1) + for index, params := range events { + event, ok := params.(map[string]any) + if !ok { + t.Fatalf("install progress event %d params type = %T, want map[string]any", index, params) + } + stage, ok := event["stage"].(string) + if !ok { + t.Fatalf("install progress event %d stage = %v, want string", index, event["stage"]) + } + stages = append(stages, stage) + switch stage { + case "done", "already-installed", "failed": + terminal = append(terminal, stage) + } + } + if len(terminal) != 1 || terminal[0] != want { + t.Fatalf("terminal install progress stages = %v, want [%s]; all stages = %v", terminal, want, stages) + } +} + func TestInstallDownloadsAllNamedArtifactsBeforeRunning(t *testing.T) { payload := []byte("artifact") server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { @@ -32,10 +75,13 @@ func TestInstallDownloadsAllNamedArtifactsBeforeRunning(t *testing.T) { } manifest := artifactInstallManifest(t, "artifact-success", marker, artifacts) executor := newTestExecutor(t, manifest) + progress := &installProgressRecorder{} + executor.emit = progress.emit if err := executor.Install(context.Background(), manifest.Engine); err != nil { t.Fatalf("install named artifacts: %v", err) } + progress.assertSingleTerminal(t, "done") data, err := os.ReadFile(marker) if err != nil { t.Fatalf("read captured install arguments: %v", err) @@ -64,11 +110,14 @@ func TestInstallRejectsBadSecondArtifactBeforeCommand(t *testing.T) { } manifest := artifactInstallManifest(t, "artifact-bad-checksum", marker, artifacts) executor := newTestExecutor(t, manifest) + progress := &installProgressRecorder{} + executor.emit = progress.emit err := executor.Install(context.Background(), manifest.Engine) if err == nil || !strings.Contains(err.Error(), "checksum mismatch") { t.Fatalf("install error = %v, want checksum mismatch", err) } + progress.assertSingleTerminal(t, "failed") if fileExists(marker) { t.Fatal("install command ran after an artifact checksum failed") } @@ -99,6 +148,8 @@ func TestInstallCancellationRemovesDownloadedArtifacts(t *testing.T) { } manifest := artifactInstallManifest(t, "artifact-cancel", marker, artifacts) executor := newTestExecutor(t, manifest) + progress := &installProgressRecorder{} + executor.emit = progress.emit ctx, cancel := context.WithCancel(context.Background()) done := make(chan error, 1) go func() { @@ -119,6 +170,7 @@ func TestInstallCancellationRemovesDownloadedArtifacts(t *testing.T) { case <-time.After(5 * time.Second): t.Fatal("cancelled install did not return") } + progress.assertSingleTerminal(t, "failed") assertNoArtifactTemps(t, manifest.Engine) } From 2794e381b26e744a35bde512b62ed56f8f3f933c Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Fri, 2 Oct 2026 23:47:31 -0700 Subject: [PATCH 54/70] fix(proxy): aggregate llama.cpp models through both list routes Signed-off-by: Sherief Farouk --- services/nvpair-proxy/engines.go | 5 ++- services/nvpair-proxy/engines_test.go | 30 +++++++++++-- services/nvpair-proxy/failover_test.go | 60 +++++++++++++++++++++++++ services/tests/llamacpp_interop_test.go | 14 ++++-- 4 files changed, 100 insertions(+), 9 deletions(-) diff --git a/services/nvpair-proxy/engines.go b/services/nvpair-proxy/engines.go index db94e705..9ad19bab 100644 --- a/services/nvpair-proxy/engines.go +++ b/services/nvpair-proxy/engines.go @@ -138,9 +138,10 @@ var lmStudioBaseRoutes = []route{ {Path: "/v1/models", Role: roleModelListOpenAIGET}, } -// llamaCPPBaseRoutes maps the facade's OpenAI-compatible model-list path to -// llama.cpp's router endpoint. The response already uses the OpenAI envelope. +// llamaCPPBaseRoutes serves both model-list paths through the fleet inventory, +// querying llama.cpp's router endpoint. Its response uses the OpenAI envelope. var llamaCPPBaseRoutes = []route{ + {Path: "/models", Role: roleModelListOpenAIGET}, {Path: "/v1/models", UpstreamPath: "/models", Role: roleModelListOpenAIGET}, } diff --git a/services/nvpair-proxy/engines_test.go b/services/nvpair-proxy/engines_test.go index feaf79b3..201a3046 100644 --- a/services/nvpair-proxy/engines_test.go +++ b/services/nvpair-proxy/engines_test.go @@ -4,6 +4,7 @@ package main import ( + "net/http" "slices" "testing" @@ -25,10 +26,6 @@ func TestLlamaCPPProfile(t *testing.T) { if !ok { t.Fatal("llamacpp profile missing") } - route, ok := profile.routeFor("GET", "/v1/models") - if !ok || route.Role != roleModelListOpenAIGET || route.upstreamPath() != "/models" { - t.Fatalf("model list route = %+v, %v", route, ok) - } if role, ok := profile.roleFor("POST", "/v1/chat/completions"); !ok || role != roleInferencePOST { t.Fatalf("chat route = %v, %v", role, ok) } @@ -40,6 +37,24 @@ func TestLlamaCPPProfile(t *testing.T) { } } +func TestLlamaCPPModelListRoutes(t *testing.T) { + profile, ok := profileFor("llamacpp") + if !ok { + t.Fatal("llamacpp profile missing") + } + for _, path := range []string{"/models", "/v1/models"} { + t.Run(path, func(t *testing.T) { + route, ok := profile.routeFor(http.MethodGet, path) + if !ok { + t.Fatalf("GET %s is not classified", path) + } + if route.Role != roleModelListOpenAIGET || route.upstreamPath() != "/models" { + t.Fatalf("GET %s route = %+v, want OpenAI model list at upstream /models", path, route) + } + }) + } +} + // Routes is a classifier, not an allowlist. handlePlain forwards every // loopback path into handleHTTP with no filtering, so a path the table does // not mention must still reach the upstream verbatim — that is how /api/show, @@ -53,6 +68,10 @@ func TestRoleForClassifiesOnlyDeclaredRoutes(t *testing.T) { if !ok { t.Fatal("lmstudio profile missing") } + llamacpp, ok := profileFor("llamacpp") + if !ok { + t.Fatal("llamacpp profile missing") + } for _, tc := range []struct { name string @@ -69,10 +88,12 @@ func TestRoleForClassifiesOnlyDeclaredRoutes(t *testing.T) { {"ollama openai list", ollama, "GET", "/v1/models", roleModelListOpenAIGET, true}, {"ollama passthrough", ollama, "POST", "/api/pull", 0, false}, {"ollama version passthrough", ollama, "GET", "/api/version", 0, false}, + {"ollama models passthrough", ollama, http.MethodGet, "/models", 0, false}, {"lmstudio chat", lmstudio, "POST", "/v1/chat/completions", roleInferencePOST, true}, {"lmstudio anthropic messages", lmstudio, "POST", "/v1/messages", roleInferencePOST, true}, {"lmstudio list", lmstudio, "GET", "/v1/models", roleModelListOpenAIGET, true}, + {"lmstudio models passthrough", lmstudio, http.MethodGet, "/models", 0, false}, // LM Studio serves no native Ollama routes, so /api/chat is not // inference for it — it is forwarded verbatim like any other path. {"lmstudio has no native routes", lmstudio, "POST", "/api/chat", 0, false}, @@ -82,6 +103,7 @@ func TestRoleForClassifiesOnlyDeclaredRoutes(t *testing.T) { // inference path would emit a workload. {"wrong method on list", ollama, "POST", "/v1/models", 0, false}, {"wrong method on inference", ollama, "GET", "/api/chat", 0, false}, + {"llamacpp models post passthrough", llamacpp, http.MethodPost, "/models", 0, false}, } { t.Run(tc.name, func(t *testing.T) { role, ok := tc.profile.roleFor(tc.method, tc.path) diff --git a/services/nvpair-proxy/failover_test.go b/services/nvpair-proxy/failover_test.go index a1505226..049e1d13 100644 --- a/services/nvpair-proxy/failover_test.go +++ b/services/nvpair-proxy/failover_test.go @@ -13,6 +13,7 @@ import ( "net/url" "strconv" "strings" + "sync/atomic" "testing" "time" ) @@ -536,6 +537,65 @@ func TestHandleHTTP_AggregatesOpenAIModelList(t *testing.T) { }) } +func TestHandleHTTP_LlamaCPPModelListsAggregateFleet(t *testing.T) { + for _, path := range []string{"/models", "/v1/models"} { + t.Run(path, func(t *testing.T) { + profile, ok := profileFor("llamacpp") + if !ok { + t.Fatal("llamacpp profile missing") + } + serve := func(body string, hits *atomic.Int32) *httptest.Server { + t.Helper() + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + hits.Add(1) + if r.Method != http.MethodGet || r.URL.Path != "/models" || r.URL.RawQuery != "scope=all" { + t.Errorf("upstream request = %s %s?%s, want GET /models?scope=all", r.Method, r.URL.Path, r.URL.RawQuery) + } + w.Header().Set("Content-Type", "application/json") + if _, err := io.WriteString(w, body); err != nil { + t.Errorf("write model list: %v", err) + } + })) + t.Cleanup(server.Close) + return server + } + var aHits, bHits atomic.Int32 + a := serve(`{"object":"list","data":[{"id":"a","owned_by":"a"},{"id":"shared","owned_by":"first"}]}`, &aHits) + b := serve(`{"object":"list","data":[{"id":"shared","owned_by":"second"},{"id":"c","owned_by":"b"}]}`, &bHits) + disc := NewDiscovery() + disc.AddManual(nodeFor(t, "a", a.URL)) + disc.AddManual(nodeFor(t, "b", b.URL)) + f := testProxy(profile, disc, profile.FacadePort).soleFacade() + f.SetSelected("a") + rec := httptest.NewRecorder() + f.handleHTTP(rec, httptest.NewRequest(http.MethodGet, path+"?scope=all", nil)) + + if rec.Code != http.StatusOK { + t.Fatalf("model list status = %d, want %d", rec.Code, http.StatusOK) + } + var got struct { + Object string `json:"object"` + Data []struct { + ID string `json:"id"` + OwnedBy string `json:"owned_by"` + } `json:"data"` + } + if err := json.Unmarshal(rec.Body.Bytes(), &got); err != nil { + t.Fatalf("decode model list: %v", err) + } + if got.Object != "list" || len(got.Data) != 3 { + t.Fatalf("model list = %+v, want list envelope with three deduplicated models", got) + } + if got.Data[0].ID != "a" || got.Data[1].ID != "shared" || got.Data[1].OwnedBy != "first" || got.Data[2].ID != "c" { + t.Fatalf("models = %+v, want a, shared(first), c", got.Data) + } + if aHits.Load() != 1 || bHits.Load() != 1 { + t.Fatalf("upstream requests: a=%d, b=%d, want one per node despite selecting a", aHits.Load(), bHits.Load()) + } + }) + } +} + func TestHandleHTTP_ModelListRemapsUpstreamPath(t *testing.T) { upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { if r.Method != http.MethodGet || r.URL.Path != "/models" || r.URL.RawQuery != "scope=all" { diff --git a/services/tests/llamacpp_interop_test.go b/services/tests/llamacpp_interop_test.go index 86f988dc..495f3dbd 100644 --- a/services/tests/llamacpp_interop_test.go +++ b/services/tests/llamacpp_interop_test.go @@ -73,8 +73,10 @@ func TestLlamaCPPFacadeUsesRouterInventoryAndExactModelIDs(t *testing.T) { client := &http.Client{Timeout: 5 * time.Second} t.Cleanup(client.CloseIdleConnections) - t.Run("remaps OpenAI model list to router inventory", func(t *testing.T) { - response, err := client.Get(fmt.Sprintf("http://127.0.0.1:%d/v1/models", proxyPort)) + getModelList := func(t *testing.T, path string) { + t.Helper() + hitsBefore := modelListHits.Load() + response, err := client.Get(fmt.Sprintf("http://127.0.0.1:%d%s", proxyPort, path)) if err != nil { t.Fatalf("get model list: %v", err) } @@ -95,10 +97,16 @@ func TestLlamaCPPFacadeUsesRouterInventoryAndExactModelIDs(t *testing.T) { } } if response.StatusCode != http.StatusOK || list.Object != "list" || - !found || modelListHits.Load() != 1 { + !found || modelListHits.Load() != hitsBefore+1 { t.Fatalf("model list status=%d body=%+v upstreamHits=%d", response.StatusCode, list, modelListHits.Load()) } + } + t.Run("remaps OpenAI model list to router inventory", func(t *testing.T) { + getModelList(t, "/v1/models") + }) + t.Run("serves router model-list alias", func(t *testing.T) { + getModelList(t, "/models") }) post := func(t *testing.T, requestedModel string) int { From 9c892b03296caa248295f550a055a37cc5f41f4e Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Fri, 2 Oct 2026 23:47:31 -0700 Subject: [PATCH 55/70] docs(proxy): document llama.cpp fleet model-list aliases Signed-off-by: Sherief Farouk --- docs/architecture.mdx | 7 ++++--- services/nvpair-proxy/README.md | 4 +++- services/nvpair-proxy/spec.md | 7 +++++-- 3 files changed, 12 insertions(+), 6 deletions(-) diff --git a/docs/architecture.mdx b/docs/architecture.mdx index 08478267..dcb05cb3 100644 --- a/docs/architecture.mdx +++ b/docs/architecture.mdx @@ -366,9 +366,10 @@ advertised owner eligible. A request whose model cannot be parsed keeps the ordinary non-model ordering. -Model listings are not routed at all. A `GET` of `/v1/models` or `/api/tags` is -fanned out to every candidate concurrently and the replies are merged, which is -why the answer is the cluster's inventory rather than one node's. +Model listings are not routed at all. A `GET` of `/v1/models`, Ollama's +`/api/tags`, or llama.cpp's `/models` is fanned out to every candidate for that +engine concurrently and the replies are merged, which is why the answer is the +cluster's inventory for that engine rather than one node's. #### Failover and Inventory Freshness diff --git a/services/nvpair-proxy/README.md b/services/nvpair-proxy/README.md index e45cfa02..87346832 100644 --- a/services/nvpair-proxy/README.md +++ b/services/nvpair-proxy/README.md @@ -101,7 +101,7 @@ and the health crash key are matched against each other, so they move together | Where PAIR relocates the engine | 11435 | 1235 | 8081 | | Standalone port, used when `port` is omitted | 11435 | 1234 | 8080 | | Persisted-port file (declared, not derived) | `proxy-port.json` | `lmstudio-proxy-port.json` | `llamacpp-proxy-port.json` | -| Model-list routes | `GET /api/tags` (native), `GET /v1/models` (OpenAI) | `GET /v1/models` (OpenAI) | `GET /v1/models` (OpenAI; upstream `/models`) | +| Model-list routes | `GET /api/tags` (native), `GET /v1/models` (OpenAI) | `GET /v1/models` (OpenAI) | `GET /models`, `GET /v1/models` (OpenAI; both query upstream `/models`) | | Inference routes | `/api/generate`, `/api/chat`, `/api/embeddings`, `/api/embed`, plus the OpenAI and Anthropic Messages sets | `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings`, `/v1/messages` | `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings` | | Model naming | untagged means `:latest`, so `llama3` and `llama3:latest` are one model | identifiers compared byte for byte | identifiers compared byte for byte | @@ -118,6 +118,8 @@ clients use `8080`, while the managed `llama-server` stays on `8081`. A facade listens on its enabled port and forwards incoming requests to the currently active node — except the model-list routes, which are queried across every candidate node concurrently and merged into one de-duplicated inventory. +On the llama.cpp facade, `GET /models` and `GET /v1/models` return the same +fleet inventory across llama.cpp candidates, even when a node is selected. Point your client at the proxy and it handles routing. When the broker supplies `aliasAddresses`, the facade reserves that diff --git a/services/nvpair-proxy/spec.md b/services/nvpair-proxy/spec.md index 457e5f21..0ab6722f 100644 --- a/services/nvpair-proxy/spec.md +++ b/services/nvpair-proxy/spec.md @@ -178,8 +178,11 @@ For a model-bearing inference request: An ineligible manual selection cannot override the capability gate, and failover never broadens to an excluded node. -The llama.cpp facade exposes the OpenAI-compatible `GET /v1/models` route and -remaps it to the router's `GET /models`; its inference routes remain `/v1/*`. +The llama.cpp facade exposes `GET /models` and `GET /v1/models` as aliases for +the fleet's llama.cpp inventory. Both query every candidate's `GET /models` +concurrently and merge duplicate model ids into an OpenAI list envelope +(`{"object":"list","data":[...]}`). A selected node affects candidate ordering, +not inventory scope. Its inference routes remain `/v1/*`. ### 5.1 Retry bounds From b055329886d78efcb8d9e3d00e99d54e55b5521d Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Sat, 3 Oct 2026 00:42:09 -0700 Subject: [PATCH 56/70] fix(engine-manager): reject download installs without run Signed-off-by: Sherief Farouk --- services/nvpair-engine-manager/registry.go | 3 + .../nvpair-engine-manager/registry_test.go | 79 +++++++++++++++++++ 2 files changed, 82 insertions(+) diff --git a/services/nvpair-engine-manager/registry.go b/services/nvpair-engine-manager/registry.go index c1025385..c1c1a501 100644 --- a/services/nvpair-engine-manager/registry.go +++ b/services/nvpair-engine-manager/registry.go @@ -677,6 +677,9 @@ func (p *Platform) validate(key string) error { if p.Install.Fetch != nil && hasArtifacts { return fmt.Errorf("platform %q: install.fetch and install.artifacts are mutually exclusive", key) } + if (p.Install.Fetch != nil || hasArtifacts) && len(p.Install.Run) == 0 { + return fmt.Errorf("platform %q: install.run is required when install.fetch or install.artifacts is present", key) + } if len(p.Install.Run) > 0 && p.Install.Fetch == nil && !hasArtifacts { return fmt.Errorf("platform %q: install.run requires a fetch or artifacts (the downloads the run command uses)", key) } diff --git a/services/nvpair-engine-manager/registry_test.go b/services/nvpair-engine-manager/registry_test.go index eccf40d2..26859d85 100644 --- a/services/nvpair-engine-manager/registry_test.go +++ b/services/nvpair-engine-manager/registry_test.go @@ -156,6 +156,51 @@ func TestValidateAcceptsNamedInstallArtifacts(t *testing.T) { } } +func TestValidateRejectsDownloadInstallWithoutRun(t *testing.T) { + test := func(name string, install *Install) { + t.Run(name, func(t *testing.T) { + m := validManifest() + p := m.Platforms["linux/amd64"] + p.Install = install + m.Platforms["linux/amd64"] = p + + err := m.Validate() + if err == nil { + t.Fatal("download install without run accepted") + } + const want = `platform "linux/amd64": install.run is required when install.fetch or install.artifacts is present` + if err.Error() != want { + t.Fatalf("validation error = %q, want %q", err, want) + } + }) + } + + test("fetch with omitted run", &Install{ + Fetch: &Fetch{URL: "https://example/installer.zip"}, + }) + test("fetch with empty run", &Install{ + Fetch: &Fetch{URL: "https://example/installer.zip"}, + Run: []string{}, + }) + test("artifacts with omitted run", &Install{ + Artifacts: validInstallArtifacts(), + }) + test("artifacts with empty run", &Install{ + Artifacts: validInstallArtifacts(), + Run: []string{}, + }) +} + +func TestValidateAcceptsScriptOnlyInstall(t *testing.T) { + m := validManifest() + p := m.Platforms["linux/amd64"] + p.Install = &Install{Script: []string{"sh", "installer.sh"}} + m.Platforms["linux/amd64"] = p + if err := m.Validate(); err != nil { + t.Fatalf("script-only install rejected: %v", err) + } +} + func TestValidateRejectsArtifactPlaceholderFromAnotherPlatform(t *testing.T) { m := validManifest() setInstallArtifacts(&m, validInstallArtifacts()) @@ -417,6 +462,40 @@ func TestLoadRegistryOverride(t *testing.T) { } } +func TestLoadOverrideDirRejectsEmptyInstallRun(t *testing.T) { + reg := NewRegistry() + if err := reg.LoadFS(bundledManifests, "manifests"); err != nil { + t.Fatalf("load bundled manifests: %v", err) + } + base, ok := reg.Get("ollama") + if !ok { + t.Fatal("bundled ollama manifest missing") + } + baseRun := base.Platforms["linux/amd64"].Install.Run + if len(baseRun) == 0 { + t.Fatal("bundled ollama install.run is empty") + } + dir := t.TempDir() + const override = `{ + "engine": "ollama", + "display_name": "Invalid override", + "platforms": {"linux/amd64": {"install": {"run": []}}} +}` + if err := os.WriteFile(filepath.Join(dir, "ollama.json"), []byte(override), 0o644); err != nil { + t.Fatalf("write override: %v", err) + } + if err := reg.LoadOverrideDir(dir); err != nil { + t.Fatalf("load override directory: %v", err) + } + got, ok := reg.Get("ollama") + if !ok || got != base { + t.Fatalf("invalid override replaced bundled manifest: got %+v, want original manifest", got) + } + if !slices.Equal(got.Platforms["linux/amd64"].Install.Run, baseRun) { + t.Fatal("invalid override changed the bundled install.run") + } +} + func TestLoadRegistryRejectsInvalidFile(t *testing.T) { dir := t.TempDir() if err := os.WriteFile(filepath.Join(dir, "broken.json"), []byte(`{"engine":"x"}`), 0o644); err != nil { From ec5a7829f79fc8c56c678d364310cdfcd65d096a Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Sat, 3 Oct 2026 00:42:10 -0700 Subject: [PATCH 57/70] docs(engine-manager): require run for download installs Signed-off-by: Sherief Farouk --- services/nvpair-engine-manager/MANIFEST.md | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/services/nvpair-engine-manager/MANIFEST.md b/services/nvpair-engine-manager/MANIFEST.md index 2f2040b6..c7282520 100644 --- a/services/nvpair-engine-manager/MANIFEST.md +++ b/services/nvpair-engine-manager/MANIFEST.md @@ -141,7 +141,7 @@ and recovery. Editing `args`/`start` directly remains trusted manifest authoring |---|---|---|---| | `fetch.url` | string | when `fetch` present | Download URL — **HTTPS** (plain `http` only from loopback). | | `fetch.sha256` | string | no | Hex SHA-256. When set, the download is verified against it **before** `run` executes; when omitted, the fetch is HTTPS-only and runs with a loud "unpinned" warning (the same weaker guarantee as `script`). Pin it for any real release. | -| `run` | string[] | no | Argv to execute after download (e.g. run the installer, extract the archive). Placeholders resolved; OS env refs expanded. The child also receives exact paths in `NVPAIR_INSTALL_DIR`, `NVPAIR_INSTALL_DOWNLOAD`, and `NVPAIR_INSTALL_DOWNLOAD_` so commands that reparse argv can avoid shell quoting. Requires a `fetch` or `artifacts`. | +| `run` | string[] | when `fetch` or nonempty `artifacts` present | Nonempty argv to execute after download (e.g. run the installer, extract the archive). Placeholders resolved; OS env refs expanded. The child also receives exact paths in `NVPAIR_INSTALL_DIR`, `NVPAIR_INSTALL_DOWNLOAD`, and `NVPAIR_INSTALL_DOWNLOAD_` so commands that reparse argv can avoid shell quoting. Requires a `fetch` or `artifacts`. | | `script` | string[] | no | **Escape hatch** for vendors that only ship a script installer. Runs **without** checksum verification (logged as unpinned) and replaces `fetch`+`run`. Prefer `fetch`+`run` whenever the vendor publishes a script or artifact: download it first, then execute the local file. **Make failures loud:** a piped bootstrap such as `curl … \| bash` can mask a failed fetch, while a separate fetch prevents the run and reports the error. | | `mode` | string | no | `"user"` (default) or `"admin"`. The runner **refuses** `"admin"` (engine-manager is user-mode only); it is a deliberate, flagged exception, not a default. | @@ -322,8 +322,10 @@ substituted into `http.path`, which templates only `{port}`. A manifest is rejected at load (with a specific message) when: a required field is missing, `manifest_version` is unsupported, a platform key isn't `"/"`, `runtime.bin` is empty in process mode (or -`runtime.start` is empty in command mode), `install.run` has no `fetch`, -`install.script` is combined with `fetch`/`run`, `install.mode` or +`runtime.start` is empty in command mode), `install.run` has neither `fetch` +nor nonempty `artifacts`, `fetch` or nonempty `artifacts` is present without +nonempty `install.run`, `install.script` is combined with +`fetch`/`artifacts`/`run`, `install.mode` or `runtime.mode` is invalid, an action sets none or more than one of `http`/`cmd`/`remove_path`, a `remove_path` action omits `root` or `path`, `http.params_in` is not `body` or `query`, a `result` is set without From ab262b266949907bed881a39d47e9cba08413a2d Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Sat, 3 Oct 2026 01:20:48 -0700 Subject: [PATCH 58/70] fix(broker): validate proxy port changes through settings for every engine Signed-off-by: Sherief Farouk --- services/nvpair-ui-broker/broker.go | 6 +- services/nvpair-ui-broker/engineproxy.go | 6 + .../nvpair-ui-broker/enginesettings_test.go | 18 +- services/nvpair-ui-broker/settingsport.go | 2 +- .../nvpair-ui-broker/settingsport_test.go | 186 ++++++++++++++++++ 5 files changed, 208 insertions(+), 10 deletions(-) create mode 100644 services/nvpair-ui-broker/settingsport_test.go diff --git a/services/nvpair-ui-broker/broker.go b/services/nvpair-ui-broker/broker.go index e83a9d25..aa83ec44 100644 --- a/services/nvpair-ui-broker/broker.go +++ b/services/nvpair-ui-broker/broker.go @@ -3240,10 +3240,6 @@ func (b *Broker) handleMessage(msg *Message) { case "engine:set-port": go b.handleSettingsPortRPC(msg, "") - case "ollama-proxy:set-port": - go b.handleSettingsPortRPC(msg, "ollama") - case "lmstudio-proxy:set-port": - go b.handleSettingsPortRPC(msg, "lmstudio") case "workloads:subscribe": b.workloadsMu.Lock() @@ -3331,7 +3327,7 @@ func (b *Broker) handleMessage(msg *Message) { default: // Any remaining method under an engine's : prefix is // relayed verbatim to that engine's proxy (the reserved broker-local - // ones — get-status and the subscription methods — are handled by + // ones — get-status, set-port and the subscription methods — are handled by // the profile-driven block above). This makes the broker a thin pass-through // for each proxy's whole control plane without enumerating methods. // diff --git a/services/nvpair-ui-broker/engineproxy.go b/services/nvpair-ui-broker/engineproxy.go index 036370ca..975882f7 100644 --- a/services/nvpair-ui-broker/engineproxy.go +++ b/services/nvpair-ui-broker/engineproxy.go @@ -572,6 +572,12 @@ func (b *Broker) setEngineProxySubscribed(p engineProxyProfile, subscribed bool) // rather than the proxy child. It reports whether method was handled. func (b *Broker) handleEngineProxyBrokerRequest(profile engineProxyProfile, method string, msg *Message) bool { switch method { + case "set-port": + // Settings application round-trips through worker readers, so it + // must not block the broker's JSON-RPC read pump. + go b.handleSettingsPortRPC(msg, profile.Name) + return true + case "get-status": var result ProxyStatusResult if proxy := b.engineProxyHandle(profile); proxy != nil { diff --git a/services/nvpair-ui-broker/enginesettings_test.go b/services/nvpair-ui-broker/enginesettings_test.go index 1344787b..6dc1e892 100644 --- a/services/nvpair-ui-broker/enginesettings_test.go +++ b/services/nvpair-ui-broker/enginesettings_test.go @@ -25,6 +25,7 @@ import ( type settingsHarness struct { b *Broker applies atomic.Int32 + proxyRebinds atomic.Int32 fail atomic.Bool failBeforeStop atomic.Bool loseProxyOnStop atomic.Bool @@ -38,6 +39,11 @@ type settingsHarness struct { } func newSettingsHarness(t *testing.T) *settingsHarness { + t.Helper() + return newSettingsHarnessForEngine(t, "ollama") +} + +func newSettingsHarnessForEngine(t *testing.T, engine string) *settingsHarness { t.Helper() ports := make([]int, 4) listeners := []net.Listener{} @@ -81,14 +87,18 @@ func newSettingsHarness(t *testing.T) *settingsHarness { var p struct { Port int `json:"port"` } - _ = json.Unmarshal(msg.Params, &p) + if err := json.Unmarshal(msg.Params, &p); err != nil { + t.Errorf("decode proxy port request: %v", err) + return + } + h.proxyRebinds.Add(1) proxy.readyMu.Lock() proxy.facadeState[engine] = proxyFacadeState{ready: true, port: p.Port} proxy.readyMu.Unlock() _ = proxyCodec.Respond(msg.ID, map[string]int{"port": p.Port}) } }() - launch := settings.LaunchState{Engine: "ollama", ServerPort: ports[0], EffectivePort: ports[0], LaunchText: "--fixture-option", Running: true, Editable: true, Format: "pair-arguments-v1"} + launch := settings.LaunchState{Engine: engine, ServerPort: ports[0], EffectivePort: ports[0], LaunchText: "--fixture-option", Running: true, Editable: true, Format: "pair-arguments-v1"} go func() { for { msg, err := codec.Read() @@ -130,9 +140,9 @@ func newSettingsHarness(t *testing.T) *settingsHarness { if h.failBeforeStop.Load() { if h.loseProxyOnStop.Load() { proxy.readyMu.Lock() - state := proxy.facadeState["ollama"] + state := proxy.facadeState[engine] state.ready = false - proxy.facadeState["ollama"] = state + proxy.facadeState[engine] = state proxy.readyMu.Unlock() } _ = codec.RespondError(msg.ID, -32000, "stop failure") diff --git a/services/nvpair-ui-broker/settingsport.go b/services/nvpair-ui-broker/settingsport.go index c1d11105..763c5065 100644 --- a/services/nvpair-ui-broker/settingsport.go +++ b/services/nvpair-ui-broker/settingsport.go @@ -12,7 +12,7 @@ import ( ) // handleSettingsPortRPC serves the port-only RPCs — engine:set-port, -// proxy:set-port, and lmstudio-proxy:set-port — which nvpair-tui calls to move +// -proxy:set-port — which nvpair-tui calls to move // a single port without rendering the full launch settings form. They keep // their own narrow request and response shapes, but run through the same // authoritative settings operation as the desktop editor, so a port change diff --git a/services/nvpair-ui-broker/settingsport_test.go b/services/nvpair-ui-broker/settingsport_test.go new file mode 100644 index 00000000..538a3898 --- /dev/null +++ b/services/nvpair-ui-broker/settingsport_test.go @@ -0,0 +1,186 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "context" + "encoding/json" + "net" + "os" + "strings" + "testing" + "time" + + settings "nvpair-shared/enginesettings" +) + +func callBrokerPortRequest(t *testing.T, b *Broker, method string, params json.RawMessage) *Message { + t.Helper() + client, server := net.Pipe() + t.Cleanup(func() { _ = client.Close(); _ = server.Close() }) + if err := client.SetReadDeadline(time.Now().Add(5 * time.Second)); err != nil { + t.Fatalf("set response deadline: %v", err) + } + b.codec = NewCodec(server) + id := json.RawMessage(`1`) + go b.handleMessage(&Message{JSONRPC: "2.0", ID: &id, Method: method, Params: params}) + codec := NewCodec(client) + for { + response, err := codec.Read() + if err != nil { + t.Fatalf("read %s response: %v", method, err) + } + if response.IsNotification() { + continue + } + if response.ID == nil || string(*response.ID) != string(id) { + t.Fatalf("unexpected response ID: %+v", response) + } + return response + } +} + +func TestBrokerProxySetPortRejectsInvalidPorts(t *testing.T) { + for _, profile := range engineProxyProfiles { + t.Run(profile.Name, func(t *testing.T) { + for _, tc := range []struct{ name, params string }{ + {"malformed parameters", `{`}, + {"missing port", `{}`}, + {"zero port", `{"port":0}`}, + {"negative port", `{"port":-1}`}, + {"oversized port", `{"port":65536}`}, + } { + t.Run(tc.name, func(t *testing.T) { + response := callBrokerPortRequest(t, &Broker{}, profile.ComponentName()+":set-port", json.RawMessage(tc.params)) + if response.Error == nil || response.Error.Code != -32602 || response.Error.Message != "port must be between 1 and 65535" { + t.Fatalf("response = %+v, want invalid-port error", response) + } + }) + } + }) + } +} + +func TestBrokerProxySetPortRejectsInheritedAlias(t *testing.T) { + for _, profile := range engineProxyProfiles { + t.Run(profile.Name, func(t *testing.T) { + b := &Broker{} + b.setOllamaHostAlias(ollamaHostAlias{Port: 11433}) + response := callBrokerPortRequest(t, b, profile.ComponentName()+":set-port", json.RawMessage(`{"port":11433}`)) + if response.Error == nil || response.Error.Code != -32000 || !strings.Contains(response.Error.Message, "OLLAMA_HOST proxy alias") { + t.Fatalf("response = %+v, want alias-port rejection", response) + } + }) + } +} + +func TestBrokerLlamaCPPProxySetPortRejectsConflicts(t *testing.T) { + test := func(name string, port func(*testing.T, *settingsHarness, settings.Snapshot) int) { + t.Run(name, func(t *testing.T) { + h := newSettingsHarnessForEngine(t, "llamacpp") + before, err := h.b.getEngineSettings(context.Background(), settings.Request{Engine: "llamacpp"}, "") + if err != nil { + t.Fatalf("read initial settings: %v", err) + } + target := port(t, h, before) + response := callBrokerPortRequest(t, h.b, "llamacpp-proxy:set-port", settingsJSON(map[string]int{"port": target})) + if response.Error == nil || response.Error.Code != -32000 || response.Error.Message != "resolve settings errors and port conflicts before applying" { + t.Fatalf("error = %+v, want settings-conflict rejection", response.Error) + } + if h.applies.Load() != 0 || h.proxyRebinds.Load() != 0 { + t.Fatal("rejected port request changed runtime") + } + h.b.engineConfigMu.Lock() + after := h.b.engineSettings["llamacpp"].Snapshot + h.b.engineConfigMu.Unlock() + if after.Settings != before.Settings || after.Revision != before.Revision { + t.Fatalf("rejection changed desired settings: before=%+v after=%+v", before, after) + } + }) + } + test("PAIR service port", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int { + return engineControlPort + }) + test("same engine server port", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int { + return before.Settings.ServerPort + }) + test("another configured engine", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int { + h.otherEnginePort.Store(25001) + return 25001 + }) + test("another proxy listener", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int { + _, port := h.b.getProxy().Status("ollama") + return port + }) + test("occupied listener", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int { + ln, err := net.Listen("tcp", ":0") + if err != nil { + t.Fatalf("bind occupied port: %v", err) + } + t.Cleanup(func() { _ = ln.Close() }) + return ln.Addr().(*net.TCPAddr).Port + }) +} + +func TestBrokerLlamaCPPProxySetPortPersistsOnlyRequestedFacade(t *testing.T) { + h := newSettingsHarnessForEngine(t, "llamacpp") + before, err := h.b.getEngineSettings(context.Background(), settings.Request{Engine: "llamacpp"}, "") + if err != nil { + t.Fatalf("read initial settings: %v", err) + } + otherPorts := make(map[string]int) + for _, engine := range []string{"ollama", "lmstudio"} { + _, otherPorts[engine] = h.b.getProxy().Status(engine) + } + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatalf("allocate target port: %v", err) + } + target := ln.Addr().(*net.TCPAddr).Port + if err := ln.Close(); err != nil { + t.Fatalf("release target port: %v", err) + } + // The namespace determines the engine, even if parameters name another. + response := callBrokerPortRequest(t, h.b, "llamacpp-proxy:set-port", settingsJSON(map[string]any{"engine": "ollama", "port": target})) + if response.Error != nil { + t.Fatalf("set proxy port: %+v", response.Error) + } + var result struct { + Port int `json:"port"` + } + if err := json.Unmarshal(response.Result, &result); err != nil { + t.Fatalf("decode port result: %v", err) + } + if result.Port != target || h.proxyRebinds.Load() != 1 { + t.Fatalf("port=%d rebinds=%d, want port=%d and one rebind", result.Port, h.proxyRebinds.Load(), target) + } + ready, actual := h.b.getProxy().Status("llamacpp") + if !ready || actual != target { + t.Fatalf("llama.cpp facade ready=%v port=%d, want %d", ready, actual, target) + } + for engine, want := range otherPorts { + _, actual := h.b.getProxy().Status(engine) + if actual != want { + t.Fatalf("%s facade moved from %d to %d", engine, want, actual) + } + } + path, err := h.b.engineSettingsPath() + if err != nil { + t.Fatalf("resolve journal path: %v", err) + } + data, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read journal: %v", err) + } + var records map[string]*engineSettingsRecord + if err := json.Unmarshal(data, &records); err != nil { + t.Fatalf("decode journal: %v", err) + } + record := records["llamacpp"] + want := before.Settings + want.ProxyPort = target + if record == nil || !record.Explicit || record.Snapshot.Settings != want || record.Snapshot.Phase != "succeeded" || record.Snapshot.Revision != before.Revision+1 { + t.Fatalf("persisted record = %+v, want successful explicit proxy-port change", record) + } +} From d99fec97f23ce346b37cbd4130a330d506caabd5 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Sat, 3 Oct 2026 01:20:48 -0700 Subject: [PATCH 59/70] test(broker): cover llama.cpp port validation across processes Signed-off-by: Sherief Farouk --- services/tests/llamacpp_interop_test.go | 32 +++++++++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/services/tests/llamacpp_interop_test.go b/services/tests/llamacpp_interop_test.go index 495f3dbd..46684834 100644 --- a/services/tests/llamacpp_interop_test.go +++ b/services/tests/llamacpp_interop_test.go @@ -29,6 +29,38 @@ func TestLlamaCPPProxyIsIncludedInBrokerDefaults(t *testing.T) { } } +func TestBrokerLlamaCPPProxySetPortRejectsInvalidPorts(t *testing.T) { + stdin, msgs, stderr, cleanup := startBrokerWith(t, + "--proxy-path", proxyBin, "--proxy-engines", "llamacpp", + ) + t.Cleanup(cleanup) + go func() { + for range stderr { + } + }() + waitForMethod(t, msgs, "app:ready", 10*time.Second) + for i, tc := range []struct{ name, params string }{ + {"missing port", `{}`}, + {"zero port", `{"port":0}`}, + {"negative port", `{"port":-1}`}, + {"oversized port", `{"port":65536}`}, + } { + t.Run(tc.name, func(t *testing.T) { + id := 7300 + i + if _, err := fmt.Fprintf(stdin, `{"jsonrpc":"2.0","id":%d,"method":"llamacpp-proxy:set-port","params":%s}`+"\n", id, tc.params); err != nil { + t.Fatalf("write proxy port request: %v", err) + } + response := waitForResponse(t, msgs, 10*time.Second) + if response.ID == nil || string(*response.ID) != fmt.Sprint(id) { + t.Fatalf("unexpected response ID: %+v", response) + } + if response.Error == nil || response.Error.Code != -32602 || response.Error.Message != "port must be between 1 and 65535" { + t.Fatalf("error = %+v, want broker invalid-port rejection", response.Error) + } + }) + } +} + func TestLlamaCPPFacadeUsesRouterInventoryAndExactModelIDs(t *testing.T) { const model = "org/router-model-GGUF:Q4_K_M" var modelListHits atomic.Int32 From d09d0a70903a3944612239754a5482ca052c5ec6 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Sat, 3 Oct 2026 01:20:48 -0700 Subject: [PATCH 60/70] docs(broker): document settings protection for all proxy port setters Signed-off-by: Sherief Farouk --- desktop/docs/services-api.md | 4 ---- services/nvpair-ui-broker/README.md | 15 +++++++++------ 2 files changed, 9 insertions(+), 10 deletions(-) diff --git a/desktop/docs/services-api.md b/desktop/docs/services-api.md index bc35d1d0..454e4068 100644 --- a/desktop/docs/services-api.md +++ b/desktop/docs/services-api.md @@ -47,8 +47,6 @@ - ⚠️ nvpair-ui-broker → engine:set-reserved-port - ⚠️ nvpair-ui-broker → engine:unsubscribe - ⚠️ nvpair-ui-broker → internal:set-reserved-port -- ⚠️ nvpair-ui-broker → lmstudio-proxy:set-port -- ⚠️ nvpair-ui-broker → ollama-proxy:set-port - ⚠️ nvpair-ui-broker → workloads:unsubscribe ### Backend binaries not listed in `modular-binaries.ts` @@ -267,14 +265,12 @@ | `engine:unsubscribe` | request (we call) | ⚠️ not called | | `errors:get-initial` | request (we call) | ✅ yes | | `internal:set-reserved-port` | request (we call) | ⚠️ not called | -| `lmstudio-proxy:set-port` | request (we call) | ⚠️ not called | | `node/add` | request (we call) | ✅ yes | | `node/discovered` | request (we call) | ✅ yes | | `node/remove` | request (we call) | ✅ yes | | `node/removed` | request (we call) | ✅ yes | | `node/updated` | request (we call) | ✅ yes | | `nodes/list` | request (we call) | ✅ yes | -| `ollama-proxy:set-port` | request (we call) | ⚠️ not called | | `ready` | request (we call) | ✅ yes | | `workloads:get-initial` | request (we call) | ✅ yes | | `workloads:remove` | request (we call) | ✅ yes | diff --git a/services/nvpair-ui-broker/README.md b/services/nvpair-ui-broker/README.md index e2b03dc9..b0833fe9 100644 --- a/services/nvpair-ui-broker/README.md +++ b/services/nvpair-ui-broker/README.md @@ -423,21 +423,24 @@ Two relay-specific error cases: - If no proxy is being supervised (or it has exited), the broker replies with error `-32000` `"ollama-proxy not available"`. - `ollama-proxy:shutdown` is **refused** with error `-32601` — the broker owns the proxy's lifecycle, so a client can't terminate it independently. Shut the broker down instead (which tears the proxy down with it). -#### `ollama-proxy:set-port` +#### `ollama-proxy:set-port` / `lmstudio-proxy:set-port` / `llamacpp-proxy:set-port` -**Intercepted, not relayed verbatim.** A port-only caller — `nvpair-tui` is the one in tree — gets to move a single port without rendering the whole launch settings form, but the change still runs through the same authoritative settings operation the desktop editor uses, so a port set from the terminal cannot diverge from one set from the UI. The broker reads the engine's current settings, substitutes the requested proxy port, and applies the result. +**Intercepted, not relayed verbatim.** For every engine profile, a port-only caller — `nvpair-tui` is the one in tree — gets to move a single port without rendering the whole launch settings form, but the change still runs through the same authoritative settings operation the desktop editor uses, so a port set from the terminal cannot diverge from one set from the UI. The namespace selects the engine. The broker reads that engine's current settings, substitutes the requested proxy port, and applies the result while preserving the server port and launch arguments. Accepted settings are saved in the settings journal. -A **requested port that is already in use is refused** with error `-32000 "port %d is already in use"`. The broker does not pick a different port on the caller's behalf: silently binding somewhere else left clients pointed at a port nothing was listening on. Retry with a free port. A request that collides with an inherited `OLLAMA_HOST` alias is refused with its own message naming that alias. The response echoes the requested port (`{"port": }`) once it is bound. +A malformed request, missing port, or port outside `1`–`65535` is refused with error `-32602 "port must be between 1 and 65535"`. A **requested port that is already in use or reserved** by a PAIR service, another configured engine/proxy, or the engine's own server port is refused with error `-32000 "resolve settings errors and port conflicts before applying"`. The broker does not pick a different port on the caller's behalf. Retry with a free port. A request that collides with an inherited `OLLAMA_HOST` alias is refused with its own message naming that alias. The response echoes the requested port (`{"port": }`) once it is bound. ```json {"jsonrpc":"2.0","id":9,"method":"ollama-proxy:set-port","params":{"port":11500}} +{"jsonrpc":"2.0","id":10,"method":"llamacpp-proxy:set-port","params":{"port":8082}} ``` -Automatic conflict resolution still exists, but only for a port the user did not just choose: when the proxy announces a (re)bound port on startup and a running engine has since taken it, the broker steers the proxy to a free port and surfaces a sticky `warning` into the errors pipeline (id `ollama-proxy:port-bumped`, `action:"none"`) explaining the move. That path **never changes an engine's port** — only the proxy is moved. Error `-32000 "ollama-proxy not available"` when no proxy is supervised. +Automatic conflict resolution still exists, but only for a port the user did not just choose: when the Ollama proxy announces a (re)bound port on startup and a running engine has since taken it, the broker steers the proxy to a free port and surfaces a sticky `warning` into the errors pipeline (id `ollama-proxy:port-bumped`, `action:"none"`) explaining the move. That path **never changes an engine's port** — only the proxy is moved. Port setters also return `-32000` if the settings operation cannot run, including when the engine manager or proxy is unavailable. #### `lmstudio-proxy:get-status` / `lmstudio-proxy:subscribe` / `lmstudio-proxy:unsubscribe` / `lmstudio-proxy:` (generic relay) -The LM Studio counterpart of the `ollama-proxy:*` surface runs the supervised `lmstudio-proxy` on compatibility port `:1234` and tracks the managed LM Studio backend on `:1235`. With managed port ownership enabled (the default), the broker identifies and moves an existing LM Studio server through engine-manager before allowing the proxy to claim `1234`; unknown owners are left untouched and force a warned proxy fallback. Disabling managed ownership preserves explicit custom backend and proxy ports. `lmstudio-proxy:get-status` reports the actual bound port; `lmstudio-proxy:subscribe` / `lmstudio-proxy:unsubscribe` opt into / out of its `lmstudio-proxy:` stream; and any other `lmstudio-proxy:` is relayed verbatim with the prefix stripped (`nodes/list`, `node/select`, `node/add-manual`, `node/remove-manual`, ...). `lmstudio-proxy:shutdown` is refused because the broker owns lifecycle ordering. Workload and error events feed the shared streams exactly as Ollama's do. +The LM Studio counterpart of the `ollama-proxy:*` surface runs the supervised `lmstudio-proxy` on compatibility port `:1234` and tracks the managed LM Studio backend on `:1235`. With managed port ownership enabled (the default), the broker identifies and moves an existing LM Studio server through engine-manager before allowing the proxy to claim `1234`; unknown owners are left untouched and force a warned proxy fallback. Disabling managed ownership preserves explicit custom backend and proxy ports. `lmstudio-proxy:get-status` reports the actual bound port; `lmstudio-proxy:subscribe` / `lmstudio-proxy:unsubscribe` opt into / out of its `lmstudio-proxy:` stream; `lmstudio-proxy:set-port` uses the settings operation described above; and other `lmstudio-proxy:` requests are relayed with the prefix translated (`nodes/list`, `node/select`, `node/add-manual`, `node/remove-manual`, ...). `lmstudio-proxy:shutdown` is refused because the broker owns lifecycle ordering. Workload and error events feed the shared streams exactly as Ollama's do. + +The same broker-local status, subscription, and port-setting methods apply to `llamacpp-proxy:*`. llama.cpp port changes use the same validation, journal, and application path; remaining methods use the generic facade relay, with `llamacpp-proxy:shutdown` refused. #### `ollama-proxy:subscribe` @@ -523,7 +526,7 @@ Opt into / out of the `engine:` stream (off by default). Acks `{ subscrib Any other `engine:*` request is forwarded to `nvpair-engine-manager` verbatim and its response relayed straight back. This covers the whole engine control plane: `engine:get-installed`, `engine:describe`, `engine:status`, `engine:install`, `engine:uninstall`, `engine:start`, `engine:stop`, `engine:restart`, `engine:action`, `engine:logs`, `engine:errors`, `engine:models`. Lifecycle ops run for minutes (reporting progress via the `engine:install-progress` / `engine:state-changed` push events), so the relay imposes **no broker-side timeout** — fire the request and watch the event stream for the outcome. Error `-32000 "engine-manager not available"` when no engine-manager is supervised. -`engine:set-port` is **not** in that generic set. Like `proxy:set-port` it is intercepted and run through the authoritative settings operation, so moving an engine's server port from a port-only caller validates and restarts exactly as the full editor does, and persists as a manifest override that survives a restart. Its response is the engine's `engine:status` result. +`engine:set-port` is **not** in that generic set. Like `-proxy:set-port` it is intercepted and run through the authoritative settings operation, so moving an engine's server port from a port-only caller validates and restarts exactly as the full editor does, and persists as a manifest override that survives a restart. Its response is the engine's `engine:status` result. #### `settings/` (generic relay) From 9a07eba59e554644da6bb6091fee9070d8b384ec Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Mon, 5 Oct 2026 11:02:59 -0700 Subject: [PATCH 61/70] fix(engine-manager): stop interrupted llama.cpp downloads Signed-off-by: Sherief Farouk --- services/nvpair-engine-manager/executor.go | 7 + services/nvpair-engine-manager/pull.go | 128 +++++++++++- .../pull_llamacpp_test.go | 194 ++++++++++++++++++ 3 files changed, 324 insertions(+), 5 deletions(-) create mode 100644 services/nvpair-engine-manager/pull_llamacpp_test.go diff --git a/services/nvpair-engine-manager/executor.go b/services/nvpair-engine-manager/executor.go index 45e84efd..2f80b16d 100644 --- a/services/nvpair-engine-manager/executor.go +++ b/services/nvpair-engine-manager/executor.go @@ -105,6 +105,11 @@ type Executor struct { // without advancing a layer/file byte count. Advancing progress refreshes // the deadline, allowing large active downloads to exceed actionTimeout. pullProgressTimeout time.Duration + // pullStartTimeout lets an asynchronous pull finish its start handshake even + // after caller cancellation, so an accepted download can still be stopped. + pullStartTimeout time.Duration + // pullCleanupTimeout bounds the inventory check and download stop together. + pullCleanupTimeout time.Duration // loadedPollInterval is the cadence of the loaded-model watcher // (loadedwatch.go), which polls each running engine's resident set and emits // engine:models-changed on change. 0 disables it. Overridable via @@ -135,6 +140,8 @@ func NewExecutor(reg *Registry, reporter *Reporter, emit func(string, any), base detectTimeout: 30 * time.Second, actionTimeout: 30 * time.Minute, pullProgressTimeout: 30 * time.Minute, + pullStartTimeout: 30 * time.Second, + pullCleanupTimeout: 5 * time.Second, loadedPollInterval: defaultLoadedPollSeconds * time.Second, loadedPoke: make(chan struct{}, 1), engines: make(map[string]*engineState), diff --git a/services/nvpair-engine-manager/pull.go b/services/nvpair-engine-manager/pull.go index 9e80260b..c84d97aa 100644 --- a/services/nvpair-engine-manager/pull.go +++ b/services/nvpair-engine-manager/pull.go @@ -201,10 +201,19 @@ func (e *Executor) PullModelStream(ctx context.Context, engine, model string, pa // pullModelLlamaCPPSSE runs llama.cpp's asynchronous router download protocol. // The SSE response must be open before POST /models because terminal events are // one-shot broadcasts; subscribing afterward can miss a fast completion. -func (e *Executor) pullModelLlamaCPPSSE(ctx context.Context, engine, model string, act Action, port int, params json.RawMessage, watchdog *pullProgressWatchdog) (json.RawMessage, error) { +func (e *Executor) pullModelLlamaCPPSSE(ctx context.Context, engine, model string, act Action, port int, params json.RawMessage, watchdog *pullProgressWatchdog) (result json.RawMessage, pullErr error) { if strings.TrimSpace(model) == "" { return nil, fmt.Errorf("pull model is required for %s", pullProgressProtocolLlamaCPPModelsSSE) } + var requested struct { + Model string `json:"model"` + } + if err := json.Unmarshal(params, &requested); err != nil { + return nil, fmt.Errorf("pull %q: decode model params: %w", model, err) + } + if requested.Model != model { + return nil, fmt.Errorf("pull %q: params.model must match the requested model", model) + } baseURL := fmt.Sprintf("http://127.0.0.1:%d", port) sseReq, err := http.NewRequestWithContext(ctx, http.MethodGet, baseURL+"/models/sse", nil) if err != nil { @@ -225,19 +234,27 @@ func (e *Executor) pullModelLlamaCPPSSE(ctx context.Context, engine, model strin if err != nil { return nil, err } - startReq, err := http.NewRequestWithContext(ctx, strings.ToUpper(act.HTTP.Method), baseURL+path, bytes.NewReader(params)) + if cause := context.Cause(ctx); cause != nil { + return nil, fmt.Errorf("pull %q: %w", model, cause) + } + // Cancelling POST /models does not cancel the router's download child. Keep + // the bounded handshake alive so cancellation cannot discard its acceptance. + startCtx, cancelStart := context.WithTimeout(context.WithoutCancel(ctx), e.pullStartTimeout) + defer cancelStart() + startReq, err := http.NewRequestWithContext(startCtx, strings.ToUpper(act.HTTP.Method), baseURL+path, bytes.NewReader(params)) if err != nil { return nil, err } startReq.Header.Set("Content-Type", "application/json") startResp, err := e.client.Do(startReq) if err != nil { - return nil, fmt.Errorf("pull %q: start download: %w", model, watchdog.resolveError(ctx, err)) + return nil, llamaCPPUnconfirmedStartError(ctx, model, err) } startData, readErr := io.ReadAll(io.LimitReader(startResp.Body, 64*1024)) startResp.Body.Close() + cancelStart() if readErr != nil { - return nil, fmt.Errorf("pull %q: read start response: %w", model, watchdog.resolveError(ctx, readErr)) + return nil, llamaCPPUnconfirmedStartError(ctx, model, readErr) } if startResp.StatusCode < 200 || startResp.StatusCode >= 300 { return nil, fmt.Errorf("pull %q: engine returned HTTP %d: %s", model, startResp.StatusCode, strings.TrimSpace(string(startData))) @@ -245,9 +262,29 @@ func (e *Executor) pullModelLlamaCPPSSE(ctx context.Context, engine, model strin var started struct { Success bool `json:"success"` } - if err := json.Unmarshal(startData, &started); err != nil || !started.Success { + if err := json.Unmarshal(startData, &started); err != nil { + return nil, llamaCPPUnconfirmedStartError(ctx, model, err) + } + if !started.Success { return nil, fmt.Errorf("pull %q: engine did not accept the download", model) } + terminal := false + defer func() { + if terminal { + return + } + if cause := context.Cause(ctx); cause != nil { + pullErr = fmt.Errorf("pull %q: %w", model, cause) + } + cleanupCtx, cancelCleanup := context.WithTimeout(context.WithoutCancel(ctx), e.pullCleanupTimeout) + defer cancelCleanup() + if err := e.stopLlamaCPPDownload(cleanupCtx, baseURL, model); err != nil { + pullErr = errors.Join(pullErr, fmt.Errorf("pull %q: could not confirm download stopped: %w", model, err)) + } + }() + if cause := context.Cause(ctx); cause != nil { + return nil, fmt.Errorf("pull %q: %w", model, cause) + } lastPct := -1 sc := bufio.NewScanner(sseResp.Body) @@ -270,9 +307,11 @@ func (e *Executor) pullModelLlamaCPPSSE(ctx context.Context, engine, model strin e.emitPullProgress(ProgressEvent{Engine: engine, Op: "pull", Stage: "downloading", Percent: pct, Message: model}) } case "download_finished": + terminal = true e.emitPullProgress(ProgressEvent{Engine: engine, Op: "pull", Stage: "success", Percent: 100, Message: model}) return json.RawMessage(startData), nil case "download_failed": + terminal = true return nil, fmt.Errorf("pull %q: llama.cpp reported download failure", model) } } @@ -282,6 +321,85 @@ func (e *Executor) pullModelLlamaCPPSSE(ctx context.Context, engine, model strin return nil, fmt.Errorf("pull %q: progress stream ended before completion", model) } +// A lost acknowledgement gives us no ownership of a router download: another +// caller may already be downloading this ID. Preserve the cause without blindly +// unloading somebody else's model. +func llamaCPPUnconfirmedStartError(ctx context.Context, model string, err error) error { + return fmt.Errorf("pull %q: download acceptance and cancellation could not be confirmed: %w", model, errors.Join(context.Cause(ctx), err)) +} + +// stopLlamaCPPDownload uses the same router as the start request. Inventory is +// checked first because a missed terminal SSE may mean the model is now loaded; +// unload would then interrupt inference rather than stop an active download. +func (e *Executor) stopLlamaCPPDownload(ctx context.Context, baseURL, model string) error { + req, err := http.NewRequestWithContext(ctx, http.MethodGet, baseURL+"/models", nil) + if err != nil { + return err + } + resp, err := e.client.Do(req) + if err != nil { + return fmt.Errorf("check download inventory: %w", err) + } + var inventory struct { + Data *[]struct { + ID string `json:"id"` + Status struct { + Value string `json:"value"` + } `json:"status"` + } `json:"data"` + } + decodeErr := json.NewDecoder(io.LimitReader(resp.Body, 8*1024*1024)).Decode(&inventory) + resp.Body.Close() + if resp.StatusCode < 200 || resp.StatusCode >= 300 { + return fmt.Errorf("check download inventory: HTTP %d", resp.StatusCode) + } + if decodeErr != nil { + return fmt.Errorf("decode download inventory: %w", decodeErr) + } + if inventory.Data == nil { + return errors.New("download inventory has no data array") + } + for _, entry := range *inventory.Data { + if entry.ID != model { + continue + } + if entry.Status.Value == "" { + return errors.New("download inventory has no model status") + } + if entry.Status.Value != "downloading" { + return nil + } + params, err := json.Marshal(map[string]string{"model": model}) + if err != nil { + return err + } + req, err := http.NewRequestWithContext(ctx, http.MethodPost, baseURL+"/models/unload", bytes.NewReader(params)) + if err != nil { + return err + } + req.Header.Set("Content-Type", "application/json") + resp, err := e.client.Do(req) + if err != nil { + return fmt.Errorf("stop download: %w", err) + } + defer resp.Body.Close() + if resp.StatusCode < 200 || resp.StatusCode >= 300 { + return fmt.Errorf("stop download: HTTP %d", resp.StatusCode) + } + var stopped struct { + Success bool `json:"success"` + } + if err := json.NewDecoder(io.LimitReader(resp.Body, 64*1024)).Decode(&stopped); err != nil { + return fmt.Errorf("decode download stop response: %w", err) + } + if !stopped.Success { + return errors.New("engine did not confirm download stopped") + } + return nil + } + return nil +} + type llamaCPPModelsEvent struct { Model string `json:"model"` Event string `json:"event"` diff --git a/services/nvpair-engine-manager/pull_llamacpp_test.go b/services/nvpair-engine-manager/pull_llamacpp_test.go new file mode 100644 index 00000000..fe571459 --- /dev/null +++ b/services/nvpair-engine-manager/pull_llamacpp_test.go @@ -0,0 +1,194 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "net/http" + "net/http/httptest" + "strings" + "sync/atomic" + "testing" + "time" +) + +const llamaCPPPullTestModel = "owner/repo:Q4_K_M" + +// The download belongs to the router, not to the SSE request. Only unload +// clears downloading, so disconnecting the stream alone cannot pass a test. +type llamaCPPPullFixture struct { + ex *Executor + server *httptest.Server + started chan struct{} + endStream chan struct{} + downloading atomic.Bool + unloads atomic.Int32 + start http.HandlerFunc + inventory http.HandlerFunc + unload http.HandlerFunc + stream http.HandlerFunc +} + +func newLlamaCPPPullFixture(t *testing.T) *llamaCPPPullFixture { + t.Helper() + f := &llamaCPPPullFixture{started: make(chan struct{}), endStream: make(chan struct{})} + mux := http.NewServeMux() + mux.HandleFunc("/models/sse", func(w http.ResponseWriter, r *http.Request) { + if f.stream != nil { + f.stream(w, r) + return + } + w.Header().Set("Content-Type", "text/event-stream") + if _, err := fmt.Fprint(w, ": ready\n\n"); err != nil { + t.Errorf("write SSE greeting: %v", err) + return + } + if err := http.NewResponseController(w).Flush(); err != nil { + t.Errorf("flush SSE greeting: %v", err) + return + } + select { + case <-r.Context().Done(): + case <-f.endStream: + } + }) + mux.HandleFunc("/models", func(w http.ResponseWriter, r *http.Request) { + switch r.Method { + case http.MethodPost: + close(f.started) + if f.start != nil { + f.start(w, r) + return + } + f.downloading.Store(true) + if _, err := fmt.Fprint(w, `{"success":true}`); err != nil { + t.Errorf("write start response: %v", err) + } + case http.MethodGet: + if f.inventory != nil { + f.inventory(w, r) + return + } + if _, err := fmt.Fprintf(w, `{"data":[{"id":%q,"status":{"value":"downloading"}},{"id":"other/model","status":{"value":"downloading"}}]}`, llamaCPPPullTestModel); err != nil { + t.Errorf("write inventory: %v", err) + } + default: + http.Error(w, "unexpected method", http.StatusMethodNotAllowed) + } + }) + mux.HandleFunc("/models/unload", func(w http.ResponseWriter, r *http.Request) { + f.unloads.Add(1) + var body struct { + Model string `json:"model"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + t.Errorf("decode stop request: %v", err) + http.Error(w, "invalid body", http.StatusBadRequest) + return + } + if r.Method != http.MethodPost || body.Model != llamaCPPPullTestModel { + t.Errorf("stop request = %s %+v, want POST for %s", r.Method, body, llamaCPPPullTestModel) + http.Error(w, "wrong download", http.StatusBadRequest) + return + } + if f.unload != nil { + f.unload(w, r) + return + } + f.downloading.Store(false) + if _, err := fmt.Fprint(w, `{"success":true}`); err != nil { + t.Errorf("write stop response: %v", err) + } + }) + f.server = httptest.NewServer(mux) + t.Cleanup(f.server.Close) + f.ex = newHTTPPullTestExecutor(t, f.server, Action{ + HTTP: &ActionHTTP{Method: http.MethodPost, Path: "/models"}, + ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE, + }) + return f +} + +func (f *llamaCPPPullFixture) pull(ctx context.Context) error { + _, err := f.ex.PullModelStream(ctx, "fake", llamaCPPPullTestModel, nil) + return err +} + +func waitLlamaCPPPullSignal(t *testing.T, signal <-chan struct{}) { + t.Helper() + select { + case <-signal: + case <-time.After(5 * time.Second): + t.Fatal("timed out waiting for pull request") + } +} + +func TestLlamaCPPPullCancellationStopsDownload(t *testing.T) { + f := newLlamaCPPPullFixture(t) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + go func() { + <-f.started + cancel() + }() + if err := f.pull(ctx); !errors.Is(err, context.Canceled) { + t.Fatalf("pull error = %v, want cancellation", err) + } + if f.downloading.Load() || f.unloads.Load() != 1 { + t.Fatalf("downloading=%t unloads=%d, want stopped with one unload", f.downloading.Load(), f.unloads.Load()) + } +} + +func TestLlamaCPPPullInactivityStopsDownload(t *testing.T) { + f := newLlamaCPPPullFixture(t) + f.ex.pullProgressTimeout = 100 * time.Millisecond + if err := f.pull(context.Background()); !errors.Is(err, errPullProgressTimeout) { + t.Fatalf("pull error = %v, want inactivity timeout", err) + } + if f.downloading.Load() || f.unloads.Load() != 1 { + t.Fatalf("downloading=%t unloads=%d, want stopped with one unload", f.downloading.Load(), f.unloads.Load()) + } +} + +func TestLlamaCPPPullStreamEOFStopsDownload(t *testing.T) { + f := newLlamaCPPPullFixture(t) + go func() { + <-f.started + close(f.endStream) + }() + if err := f.pull(context.Background()); err == nil || !strings.Contains(err.Error(), "progress stream ended before completion") { + t.Fatalf("pull error = %v, want premature stream EOF", err) + } + if f.downloading.Load() || f.unloads.Load() != 1 { + t.Fatalf("downloading=%t unloads=%d, want stopped with one unload", f.downloading.Load(), f.unloads.Load()) + } +} + +func TestLlamaCPPPullStreamReadErrorStopsDownload(t *testing.T) { + f := newLlamaCPPPullFixture(t) + f.stream = func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Length", "1000") + if _, err := fmt.Fprint(w, ": ready\n\n"); err != nil { + t.Errorf("write SSE greeting: %v", err) + return + } + if err := http.NewResponseController(w).Flush(); err != nil { + t.Errorf("flush SSE greeting: %v", err) + return + } + select { + case <-f.started: + case <-r.Context().Done(): + } + } + if err := f.pull(context.Background()); err == nil || !strings.Contains(err.Error(), "unexpected EOF") { + t.Fatalf("pull error = %v, want truncated SSE read", err) + } + if f.downloading.Load() || f.unloads.Load() != 1 { + t.Fatalf("downloading=%t unloads=%d, want stopped with one unload", f.downloading.Load(), f.unloads.Load()) + } +} From e7bd8b72be3c235b49a1aedd54f4e89e714b54e4 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Mon, 5 Oct 2026 11:25:59 -0700 Subject: [PATCH 62/70] test(engine-manager): cover llama.cpp pull cleanup races Signed-off-by: Sherief Farouk --- services/nvpair-engine-manager/e2e_test.go | 4 +- services/nvpair-engine-manager/pull.go | 7 +- .../pull_llamacpp_remote_test.go | 138 ++++++++ .../pull_llamacpp_test.go | 305 ++++++++++++++++++ 4 files changed, 450 insertions(+), 4 deletions(-) create mode 100644 services/nvpair-engine-manager/pull_llamacpp_remote_test.go diff --git a/services/nvpair-engine-manager/e2e_test.go b/services/nvpair-engine-manager/e2e_test.go index e02187a0..a86ac5a5 100644 --- a/services/nvpair-engine-manager/e2e_test.go +++ b/services/nvpair-engine-manager/e2e_test.go @@ -241,9 +241,9 @@ type e2eManager struct { stopped bool } -func startE2EManager(t *testing.T, cfg, home string) *e2eManager { +func startE2EManager(t *testing.T, cfg, home string, args ...string) *e2eManager { t.Helper() - cmd := exec.Command(managerBin) + cmd := exec.Command(managerBin, args...) cmd.Env = overrideEnv(map[string]string{"APPDATA": cfg, "LOCALAPPDATA": cfg, "XDG_CONFIG_HOME": cfg, "HOME": home}) stdin, err := cmd.StdinPipe() if err != nil { diff --git a/services/nvpair-engine-manager/pull.go b/services/nvpair-engine-manager/pull.go index c84d97aa..d7642cbe 100644 --- a/services/nvpair-engine-manager/pull.go +++ b/services/nvpair-engine-manager/pull.go @@ -260,12 +260,15 @@ func (e *Executor) pullModelLlamaCPPSSE(ctx context.Context, engine, model strin return nil, fmt.Errorf("pull %q: engine returned HTTP %d: %s", model, startResp.StatusCode, strings.TrimSpace(string(startData))) } var started struct { - Success bool `json:"success"` + Success *bool `json:"success"` } if err := json.Unmarshal(startData, &started); err != nil { return nil, llamaCPPUnconfirmedStartError(ctx, model, err) } - if !started.Success { + if started.Success == nil { + return nil, llamaCPPUnconfirmedStartError(ctx, model, errors.New("start response has no success flag")) + } + if !*started.Success { return nil, fmt.Errorf("pull %q: engine did not accept the download", model) } terminal := false diff --git a/services/nvpair-engine-manager/pull_llamacpp_remote_test.go b/services/nvpair-engine-manager/pull_llamacpp_remote_test.go new file mode 100644 index 00000000..9c6bbaca --- /dev/null +++ b/services/nvpair-engine-manager/pull_llamacpp_remote_test.go @@ -0,0 +1,138 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "bytes" + "context" + "encoding/json" + "fmt" + "net/http" + "net/http/httptest" + "path/filepath" + "strconv" + "testing" + "time" + + "nvpair-shared/clustertrust" +) + +func TestLlamaCPPPullRemoteDisconnectStopsDownload(t *testing.T) { + f := newLlamaCPPPullFixture(t) + stopped := make(chan struct{}) + f.unload = func(w http.ResponseWriter, _ *http.Request) { + f.downloading.Store(false) + if _, err := fmt.Fprint(w, `{"success":true}`); err != nil { + t.Errorf("write stop response: %v", err) + } + close(stopped) + } + s := &controlServer{exec: f.ex} + server := httptest.NewServer(http.HandlerFunc(s.handlePull)) + t.Cleanup(server.Close) + requestLlamaCPPPullAndDisconnect(t, server.Client(), server.URL, f.started) + waitLlamaCPPPullSignal(t, stopped) + if f.downloading.Load() || f.unloads.Load() != 1 { + t.Fatalf("downloading=%t unloads=%d, want remote disconnect to stop download", f.downloading.Load(), f.unloads.Load()) + } +} + +// Exercise the real ec listener, pinned mTLS, request cancellation, and cleanup +// in the compiled manager, without a real engine or model download. +func TestE2ELlamaCPPPullRemoteDisconnectStopsDownload(t *testing.T) { + f := newLlamaCPPPullFixture(t) + stopped := make(chan struct{}) + f.unload = func(w http.ResponseWriter, _ *http.Request) { + f.downloading.Store(false) + if _, err := fmt.Fprint(w, `{"success":true}`); err != nil { + t.Errorf("write stop response: %v", err) + } + close(stopped) + } + serverCert, serverKey := mintLeaf(t, "pull-server") + clientCert, clientKey := mintLeaf(t, "pull-client") + serverDir := clusterDirFor(t, serverCert, serverKey, map[string][]byte{"pull-client": clientCert}) + clientDir := clusterDirFor(t, clientCert, clientKey, map[string][]byte{"pull-server": serverCert}) + controlPort, err := freePort() + if err != nil { + t.Fatalf("allocate ec port: %v", err) + } + state, err := f.ex.state("fake") + if err != nil { + t.Fatalf("resolve fake router: %v", err) + } + manifest := testEngineManifest(fakeEngineBin) + manifest.Actions[pullModelAction] = Action{ + HTTP: &ActionHTTP{Method: http.MethodPost, Path: "/models"}, + ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE, + } + for key, platform := range manifest.Platforms { + platform.Runtime.Port = state.port + platform.Runtime.Ready.HTTP = "http://127.0.0.1:{port}/health" + platform.Runtime.Health = nil + manifest.Platforms[key] = platform + } + cfg, home := t.TempDir(), t.TempDir() + for _, dir := range []string{ + filepath.Join(cfg, "Nvidia Corporation", "Personal AI Router", "engines"), + filepath.Join(home, "Library", "Application Support", "Nvidia Corporation", "Personal AI Router", "engines"), + } { + writeE2EManifest(t, dir, manifest) + } + manager := startE2EManager(t, cfg, home, "--control-port", strconv.Itoa(controlPort), "--cluster-dir", serverDir, "--loaded-poll-interval", "0") + // The fake router is already listening. Start adopts it using its readiness + // probe; the child never spawns a real engine or another fake listener. + send(t, manager.stdin, 1, "engine:start", map[string]string{"engine": "fake"}) + var status EngineStatus + if err := json.Unmarshal(waitResult(t, manager.frames, "1", 10*time.Second), &status); err != nil { + t.Fatalf("decode start status: %v", err) + } + if !status.Running || status.Port != state.port { + t.Fatalf("engine status = %+v, want running fake router at %d", status, state.port) + } + waitPortServing(t, controlPort) + tlsConfig, ok := clustertrust.Open(clientDir).ClientTLSConfig("pull-server") + if !ok { + t.Fatal("client could not resolve pinned server TLS configuration") + } + transport := &http.Transport{TLSClientConfig: tlsConfig} + t.Cleanup(transport.CloseIdleConnections) + client := &http.Client{Transport: transport} + requestLlamaCPPPullAndDisconnect(t, client, fmt.Sprintf("https://127.0.0.1:%d", controlPort), f.started) + waitLlamaCPPPullSignal(t, stopped) + if f.downloading.Load() || f.unloads.Load() != 1 { + t.Fatalf("downloading=%t unloads=%d, want compiled manager to stop download", f.downloading.Load(), f.unloads.Load()) + } + // Shut the fixture down first: the adopted listener is owned by this test, + // and manager shutdown must not attempt to terminate the test process. + f.server.Close() + manager.stop(t) +} + +func requestLlamaCPPPullAndDisconnect(t *testing.T, client *http.Client, baseURL string, started <-chan struct{}) { + t.Helper() + body, err := json.Marshal(pullRequest{OpID: "disconnect-test", Engine: "fake", Model: llamaCPPPullTestModel}) + if err != nil { + t.Fatalf("encode remote pull: %v", err) + } + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + req, err := http.NewRequestWithContext(ctx, http.MethodPost, baseURL+controlPullPath, bytes.NewReader(body)) + if err != nil { + t.Fatalf("create remote pull: %v", err) + } + req.Header.Set("Content-Type", "application/json") + resp, err := client.Do(req) + if err != nil { + t.Fatalf("start remote pull: %v", err) + } + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK { + t.Fatalf("remote pull returned HTTP %d", resp.StatusCode) + } + waitLlamaCPPPullSignal(t, started) + if err := resp.Body.Close(); err != nil { + t.Fatalf("disconnect remote pull: %v", err) + } +} diff --git a/services/nvpair-engine-manager/pull_llamacpp_test.go b/services/nvpair-engine-manager/pull_llamacpp_test.go index fe571459..762ecb68 100644 --- a/services/nvpair-engine-manager/pull_llamacpp_test.go +++ b/services/nvpair-engine-manager/pull_llamacpp_test.go @@ -8,6 +8,7 @@ import ( "encoding/json" "errors" "fmt" + "io" "net/http" "net/http/httptest" "strings" @@ -27,6 +28,7 @@ type llamaCPPPullFixture struct { endStream chan struct{} downloading atomic.Bool unloads atomic.Int32 + inventories atomic.Int32 start http.HandlerFunc inventory http.HandlerFunc unload http.HandlerFunc @@ -37,6 +39,9 @@ func newLlamaCPPPullFixture(t *testing.T) *llamaCPPPullFixture { t.Helper() f := &llamaCPPPullFixture{started: make(chan struct{}), endStream: make(chan struct{})} mux := http.NewServeMux() + mux.HandleFunc("/health", func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusOK) + }) mux.HandleFunc("/models/sse", func(w http.ResponseWriter, r *http.Request) { if f.stream != nil { f.stream(w, r) @@ -59,6 +64,20 @@ func newLlamaCPPPullFixture(t *testing.T) *llamaCPPPullFixture { mux.HandleFunc("/models", func(w http.ResponseWriter, r *http.Request) { switch r.Method { case http.MethodPost: + data, err := io.ReadAll(r.Body) + if err != nil { + t.Errorf("read start request: %v", err) + http.Error(w, "invalid body", http.StatusBadRequest) + return + } + var body struct { + Model string `json:"model"` + } + if err := json.Unmarshal(data, &body); err != nil || body.Model != llamaCPPPullTestModel { + t.Errorf("decode start request = %+v, error %v", body, err) + http.Error(w, "wrong model", http.StatusBadRequest) + return + } close(f.started) if f.start != nil { f.start(w, r) @@ -69,6 +88,7 @@ func newLlamaCPPPullFixture(t *testing.T) *llamaCPPPullFixture { t.Errorf("write start response: %v", err) } case http.MethodGet: + f.inventories.Add(1) if f.inventory != nil { f.inventory(w, r) return @@ -192,3 +212,288 @@ func TestLlamaCPPPullStreamReadErrorStopsDownload(t *testing.T) { t.Fatalf("downloading=%t unloads=%d, want stopped with one unload", f.downloading.Load(), f.unloads.Load()) } } + +func TestLlamaCPPPullCancellationWaitsForStartAcceptance(t *testing.T) { + f := newLlamaCPPPullFixture(t) + releaseStart := make(chan struct{}, 1) + defer close(releaseStart) + f.start = func(w http.ResponseWriter, r *http.Request) { + select { + case <-releaseStart: + case <-r.Context().Done(): + t.Error("start handshake was cancelled before acceptance") + return + } + f.downloading.Store(true) + if _, err := fmt.Fprint(w, `{"success":true}`); err != nil { + t.Errorf("write accepted response: %v", err) + } + } + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + result := make(chan error, 1) + go func() { result <- f.pull(ctx) }() + waitLlamaCPPPullSignal(t, f.started) + cancel() + select { + case err := <-result: + t.Fatalf("pull returned before start acceptance: %v", err) + case <-time.After(30 * time.Millisecond): + } + // Send instead of closing so the deferred close also releases the handler if + // an assertion fails before this point. + releaseStart <- struct{}{} + select { + case err := <-result: + if !errors.Is(err, context.Canceled) || f.downloading.Load() || f.unloads.Load() != 1 { + t.Fatalf("error=%v downloading=%t unloads=%d", err, f.downloading.Load(), f.unloads.Load()) + } + case <-time.After(5 * time.Second): + t.Fatal("pull did not return after acceptance and cleanup") + } +} + +func TestLlamaCPPPullCancelledBeforeStartDoesNotDownload(t *testing.T) { + f := newLlamaCPPPullFixture(t) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + if err := f.pull(ctx); !errors.Is(err, context.Canceled) { + t.Fatalf("pull error = %v, want cancellation", err) + } + select { + case <-f.started: + t.Fatal("cancelled pull sent a start request") + default: + } + if f.unloads.Load() != 0 || f.inventories.Load() != 0 { + t.Fatal("cancelled pull attempted cleanup without starting") + } +} + +func TestLlamaCPPPullRejectsInvalidModelParamsBeforeStarting(t *testing.T) { + for _, tc := range []struct{ name, params string }{ + {"different model", `{"model":"other/model"}`}, + {"missing model", `{}`}, + {"invalid JSON", `{"model":`}, + } { + t.Run(tc.name, func(t *testing.T) { + f := newLlamaCPPPullFixture(t) + _, err := f.ex.PullModelStream(context.Background(), "fake", llamaCPPPullTestModel, json.RawMessage(tc.params)) + if err == nil { + t.Fatal("invalid model params were accepted") + } + select { + case <-f.started: + t.Fatal("invalid params sent a start request") + default: + } + if f.inventories.Load() != 0 || f.unloads.Load() != 0 { + t.Fatal("invalid params triggered cleanup") + } + }) + } +} + +func TestLlamaCPPPullRejectedStartPreservesOtherDownload(t *testing.T) { + test := func(name string, status int, body string) { + t.Run(name, func(t *testing.T) { + f := newLlamaCPPPullFixture(t) + f.downloading.Store(true) + f.start = func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(status) + if _, err := fmt.Fprint(w, body); err != nil { + t.Errorf("write rejected start: %v", err) + } + } + if err := f.pull(context.Background()); err == nil { + t.Fatal("rejected start returned success") + } + if !f.downloading.Load() || f.unloads.Load() != 0 || f.inventories.Load() != 0 { + t.Fatal("rejected start touched another download") + } + }) + } + test("HTTP rejection", http.StatusConflict, `{"error":"already exists"}`) + test("negative acknowledgement", http.StatusOK, `{"success":false}`) +} + +func TestLlamaCPPPullUnconfirmedStartDoesNotUnload(t *testing.T) { + test := func(name string, respond func(*testing.T, http.ResponseWriter, *http.Request)) { + t.Run(name, func(t *testing.T) { + f := newLlamaCPPPullFixture(t) + f.ex.pullStartTimeout = 100 * time.Millisecond + f.downloading.Store(true) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + f.start = func(w http.ResponseWriter, r *http.Request) { + cancel() + respond(t, w, r) + } + err := f.pull(ctx) + if !errors.Is(err, context.Canceled) || !strings.Contains(err.Error(), "acceptance and cancellation could not be confirmed") { + t.Fatalf("pull error = %v, want unconfirmed acceptance preserving cancellation", err) + } + if !f.downloading.Load() || f.unloads.Load() != 0 || f.inventories.Load() != 0 { + t.Fatal("unconfirmed start unloaded an unowned download") + } + }) + } + test("malformed acknowledgement", func(t *testing.T, w http.ResponseWriter, _ *http.Request) { + if _, err := fmt.Fprint(w, "not JSON"); err != nil { + t.Errorf("write malformed acknowledgement: %v", err) + } + }) + test("missing success flag", func(t *testing.T, w http.ResponseWriter, _ *http.Request) { + if _, err := fmt.Fprint(w, `{}`); err != nil { + t.Errorf("write incomplete acknowledgement: %v", err) + } + }) + test("null success flag", func(t *testing.T, w http.ResponseWriter, _ *http.Request) { + if _, err := fmt.Fprint(w, `{"success":null}`); err != nil { + t.Errorf("write null acknowledgement: %v", err) + } + }) + test("truncated acknowledgement", func(t *testing.T, w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Length", "1000") + if _, err := fmt.Fprint(w, `{"success":`); err != nil { + t.Errorf("write truncated acknowledgement: %v", err) + } + }) + test("start header timeout", func(_ *testing.T, _ http.ResponseWriter, r *http.Request) { <-r.Context().Done() }) + test("start body timeout", func(t *testing.T, w http.ResponseWriter, r *http.Request) { + if err := http.NewResponseController(w).Flush(); err != nil { + t.Errorf("flush start headers: %v", err) + return + } + <-r.Context().Done() + }) +} + +func TestLlamaCPPPullTerminalEventsSkipCleanup(t *testing.T) { + for _, event := range []string{"download_finished", "download_failed"} { + t.Run(event, func(t *testing.T) { + f := newLlamaCPPPullFixture(t) + f.stream = func(w http.ResponseWriter, r *http.Request) { + if err := http.NewResponseController(w).Flush(); err != nil { + t.Errorf("flush SSE headers: %v", err) + return + } + select { + case <-f.started: + case <-r.Context().Done(): + return + } + if _, err := fmt.Fprintf(w, "data: {\"model\":%q,\"event\":%q}\n\n", llamaCPPPullTestModel, event); err != nil { + t.Errorf("write terminal event: %v", err) + } + } + err := f.pull(context.Background()) + if (err == nil) != (event == "download_finished") { + t.Fatalf("terminal %s returned error %v", event, err) + } + if f.unloads.Load() != 0 || f.inventories.Load() != 0 { + t.Fatal("terminal event triggered cleanup") + } + }) + } +} + +func TestLlamaCPPPullCleanupPreservesCompletedModels(t *testing.T) { + for _, status := range []string{"downloaded", "loaded", "unloaded", "missing"} { + t.Run(status, func(t *testing.T) { + f := newLlamaCPPPullFixture(t) + close(f.endStream) + f.inventory = func(w http.ResponseWriter, _ *http.Request) { + f.downloading.Store(false) + body := `{"data":[{"id":"other/model","status":{"value":"downloading"}}]}` + if status != "missing" { + body = fmt.Sprintf(`{"data":[{"id":%q,"status":{"value":%q}}]}`, llamaCPPPullTestModel, status) + } + if _, err := fmt.Fprint(w, body); err != nil { + t.Errorf("write completed inventory: %v", err) + } + } + err := f.pull(context.Background()) + if err == nil || !strings.Contains(err.Error(), "progress stream ended before completion") || strings.Contains(err.Error(), "could not confirm") { + t.Fatalf("pull error = %v, want only premature SSE termination", err) + } + if f.unloads.Load() != 0 || f.inventories.Load() != 1 { + t.Fatalf("inventories=%d unloads=%d, want completed model preserved", f.inventories.Load(), f.unloads.Load()) + } + }) + } +} + +func TestLlamaCPPPullCleanupFailurePreservesCancellation(t *testing.T) { + test := func(name string, status int, body string) { + t.Run(name, func(t *testing.T) { + f := newLlamaCPPPullFixture(t) + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + f.start = func(w http.ResponseWriter, _ *http.Request) { + f.downloading.Store(true) + cancel() + if _, err := fmt.Fprint(w, `{"success":true}`); err != nil { + t.Errorf("write start response: %v", err) + } + } + f.unload = func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(status) + if _, err := fmt.Fprint(w, body); err != nil { + t.Errorf("write cleanup failure: %v", err) + } + } + err := f.pull(ctx) + if !errors.Is(err, context.Canceled) || !strings.Contains(err.Error(), "could not confirm download stopped") { + t.Fatalf("pull error = %v, want cancellation and cleanup failure", err) + } + if !f.downloading.Load() || f.unloads.Load() != 1 { + t.Fatal("failed cleanup was treated as a confirmed stop or retried") + } + }) + } + test("HTTP rejection", http.StatusServiceUnavailable, "unavailable") + test("malformed acknowledgement", http.StatusOK, "invalid JSON") + test("negative acknowledgement", http.StatusOK, `{"success":false}`) +} + +func TestLlamaCPPPullCleanupTimeoutPreservesWatchdogCause(t *testing.T) { + for _, route := range []string{"inventory", "unload"} { + t.Run(route, func(t *testing.T) { + f := newLlamaCPPPullFixture(t) + f.ex.pullProgressTimeout = 100 * time.Millisecond + f.ex.pullCleanupTimeout = 100 * time.Millisecond + hang := func(_ http.ResponseWriter, r *http.Request) { <-r.Context().Done() } + if route == "inventory" { + f.inventory = hang + } else { + f.unload = hang + } + err := f.pull(context.Background()) + if !errors.Is(err, errPullProgressTimeout) || !errors.Is(err, context.DeadlineExceeded) || !strings.Contains(err.Error(), "could not confirm download stopped") { + t.Fatalf("pull error = %v, want watchdog and cleanup timeout", err) + } + if !f.downloading.Load() { + t.Fatal("cleanup timeout was treated as a confirmed stop") + } + }) + } +} + +func TestLlamaCPPPullInvalidInventoryDoesNotConfirmStop(t *testing.T) { + for _, body := range []string{"not JSON", `{}`, `{"data":null}`, `{"data":{}}`, fmt.Sprintf(`{"data":[{"id":%q}]}`, llamaCPPPullTestModel)} { + t.Run(body, func(t *testing.T) { + f := newLlamaCPPPullFixture(t) + close(f.endStream) + f.inventory = func(w http.ResponseWriter, _ *http.Request) { + if _, err := fmt.Fprint(w, body); err != nil { + t.Errorf("write invalid inventory: %v", err) + } + } + err := f.pull(context.Background()) + if err == nil || !strings.Contains(err.Error(), "could not confirm download stopped") || f.unloads.Load() != 0 { + t.Fatalf("error=%v unloads=%d, want failed inventory validation", err, f.unloads.Load()) + } + }) + } +} From 7788f07421c9a43258c55b94406c120655af5af7 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Mon, 5 Oct 2026 11:30:27 -0700 Subject: [PATCH 63/70] docs(engine-manager): document interrupted pull cleanup Signed-off-by: Sherief Farouk --- services/nvpair-engine-manager/README.md | 16 ++++++++++++++++ services/nvpair-engine-manager/spec.md | 19 ++++++++++++++++++- 2 files changed, 34 insertions(+), 1 deletion(-) diff --git a/services/nvpair-engine-manager/README.md b/services/nvpair-engine-manager/README.md index 65313f2e..8511e2a3 100644 --- a/services/nvpair-engine-manager/README.md +++ b/services/nvpair-engine-manager/README.md @@ -100,6 +100,22 @@ watchdog, so an active download may run longer than 30 minutes; duplicate progress frames and heartbeats do not extend a stalled pull. CLI-driven pulls without structured byte progress retain the fixed 30-minute action timeout. +For llama.cpp, an accepted pull that ends before a matching `download_finished` +or `download_failed` event stops the active download. This includes caller +cancellation, remote caller disconnect, inactivity timeout, and premature SSE +termination. Cleanup checks `GET /models` and sends `POST /models/unload` only +when the exact model is still `downloading`; cached files are retained, and +models that have already completed are left alone. + +The initial `POST /models` handshake has a separate 30-second total timeout and +continues through caller cancellation so its acceptance can still be read. +The pull then waits for cleanup, which has a separate five-second budget for +the inventory check and stop request together. A failed cleanup reports that +the download could not be confirmed stopped alongside the original pull error. +If the start acknowledgement is lost or unreadable, acceptance and cancellation +are reported as unconfirmed without unloading an unowned download. The monitored +model must match `params.model`; mismatches are rejected before subscribing. + The `engine:remote-*` methods are the client half of remote engine management: engine-manager resolves the target `node` in an `ec` peer directory (fed by its own `discovery:subscribe{services:[ec]}` to the broker diff --git a/services/nvpair-engine-manager/spec.md b/services/nvpair-engine-manager/spec.md index 83e1e6b8..cdaa178b 100644 --- a/services/nvpair-engine-manager/spec.md +++ b/services/nvpair-engine-manager/spec.md @@ -56,7 +56,7 @@ extensibility story for an open-source product. - **Inference traffic** — stays with `nvpair-proxy`; this service never proxies `/api/chat` etc. - **Multi-instance per engine and an MCP server** — future-additive, not v1. - **The node's error list** — owned by `nvpair-errors`, which holds it as in-memory session state; this service only emits `errors:report` / `errors:clear`. -- Automatic cleanup of persistent llama.cpp model downloads. +- Automatic deletion of persistent llama.cpp model-cache files. ## 3. Key Use Cases - **Install an engine, user-mode**: `engine:install {engine:"ollama"}` downloads the per-OS user-scoped package (Windows/Linux standalone archive extracted into a user dir; macOS app bundle — never an elevated `Setup.exe` or `curl | sh`), checksum-verifies, extracts, re-detects. @@ -228,6 +228,23 @@ The operator stops it: `engine:stop {engine:"ollama"}` signals a process the ser The operator pulls a model: `engine:action {engine:"ollama", action:"pull_model", params:{name:"llama3.2"}}` issues the manifest-declared `POST 127.0.0.1:{port}/api/pull`. Because the action is `pull_model`, the request is routed through the streaming pull path (not the buffered `engine:action` reader): each `/api/pull` status line is emitted as an `engine:pull-progress` notification — so a local pull shows live download progress just like a remote pull's `engine:remote-progress` — and the request settles with the pull's terminal result line. Frames are coalesced (only a change in `stage` or `percent` is emitted) so a chatty engine that streams many byte-progress lines per layer doesn't flood subscribers. Streaming Ollama and llama.cpp pulls use a 30-minute inactivity watchdog that is refreshed only when a layer/file completed-byte count advances, allowing active downloads to exceed 30 minutes without letting duplicate progress or heartbeat frames keep a stalled pull alive. The engine's terminal `{"status":"success"}` surfaces as a `stage:"success"` frame; a **failed** pull emits a terminal `stage:"error", percent:-1, message:` frame in addition to the JSON-RPC error, so a UI whose synchronous call already timed out on a long download still converges off "pulling". A CLI-driven pull (LM Studio's `lms get`) has no line-level progress, so it emits one `stage:"pulling"` marker and retains the fixed 30-minute action timeout before returning the command's result. On `shutdown` (or stdin EOF) the service stops every running engine first, so none are orphaned. +For the `llamacpp-models-sse` adapter, the monitored model must match +`params.model`. Subscribe before starting, and bound the `POST /models` +acknowledgement by a separate 30-second total timeout that survives caller +cancellation. Once accepted, any exit before a matching `download_finished` or +`download_failed` event performs cleanup before returning: caller cancellation, +remote disconnect, inactivity timeout, SSE read failure, and premature SSE EOF +all take this path. Using the same captured router URL and a fresh five-second +context, query `GET /models` and send `POST /models/unload` with the exact model +only if its status is still `downloading`. Missing or completed models need no +stop request, and persistent cache files are retained. Validate the unload +acknowledgement (`success:true`); join cleanup failures to the original error +without losing cancellation or inactivity causes. A lost or malformed start +acknowledgement means acceptance and cancellation cannot be confirmed: do not +unload a download without confirmed ownership. Neither terminal SSE event +triggers cleanup. These operations retain the existing JSON-RPC and progress +payloads and do not terminate the entire engine when cleanup fails. + ## 15. Current integration / wiring The engine manager builds standalone (`go build ./...`) and is tested end to From f9678e8ae4862e8cf207e3f46db27aab649ce568 Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Mon, 5 Oct 2026 11:43:00 -0700 Subject: [PATCH 64/70] refactor(broker): remove prepositioned engine ownership Signed-off-by: Sherief Farouk --- services/nvpair-ui-broker/advertiser.go | 12 ++-- services/nvpair-ui-broker/advertiser_test.go | 18 ++--- services/nvpair-ui-broker/broker.go | 35 +++++----- ...onedproxy_test.go => enginefacade_test.go} | 40 +++++------ services/nvpair-ui-broker/engineproxy.go | 66 ++++++++----------- services/nvpair-ui-broker/engineproxy_test.go | 6 +- .../enginesettings_review_test.go | 2 +- services/nvpair-ui-broker/proxyport.go | 5 +- 8 files changed, 80 insertions(+), 104 deletions(-) rename services/nvpair-ui-broker/{prepositionedproxy_test.go => enginefacade_test.go} (74%) diff --git a/services/nvpair-ui-broker/advertiser.go b/services/nvpair-ui-broker/advertiser.go index dcb0aa0a..11e864ba 100644 --- a/services/nvpair-ui-broker/advertiser.go +++ b/services/nvpair-ui-broker/advertiser.go @@ -188,25 +188,25 @@ func (b *Broker) reconcileAdvertiseLMStudio(client *http.Client) { } } -// runAutoAdvertisePrepositioned reconciles an ungated engine whose backend is -// fixed on the port recorded in its runtime profile. -func (b *Broker) runAutoAdvertisePrepositioned(ctx context.Context, profile engineProxyProfile) { +// runAutoAdvertiseEngine reconciles an engine using the configured port +// recorded in its runtime profile, without compatibility-port reconciliation. +func (b *Broker) runAutoAdvertiseEngine(ctx context.Context, profile engineProxyProfile) { client := &http.Client{Timeout: 2 * time.Second} ticker := time.NewTicker(autoAdvertiseInterval) defer ticker.Stop() - b.reconcileAdvertisePrepositioned(profile, client) + b.reconcileAdvertiseEngine(profile, client) for { select { case <-ctx.Done(): return case <-ticker.C: - b.reconcileAdvertisePrepositioned(profile, client) + b.reconcileAdvertiseEngine(profile, client) } } } -func (b *Broker) reconcileAdvertisePrepositioned(profile engineProxyProfile, client *http.Client) { +func (b *Broker) reconcileAdvertiseEngine(profile engineProxyProfile, client *http.Client) { b.engineConfigMu.Lock() defer b.engineConfigMu.Unlock() diff --git a/services/nvpair-ui-broker/advertiser_test.go b/services/nvpair-ui-broker/advertiser_test.go index 8b42a06c..b2435ce6 100644 --- a/services/nvpair-ui-broker/advertiser_test.go +++ b/services/nvpair-ui-broker/advertiser_test.go @@ -111,7 +111,7 @@ func TestLMStudioFallbackDoesNotOverwriteKnownBackend(t *testing.T) { } } -func TestPrepositionedAdvertiserTracksBackendHealth(t *testing.T) { +func TestEngineAdvertiserTracksEngineHealth(t *testing.T) { backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { w.WriteHeader(http.StatusOK) })) @@ -129,14 +129,14 @@ func TestPrepositionedAdvertiserTracksBackendHealth(t *testing.T) { proxyPort++ } - profile := testPrepositionedProfile() + profile := testDefaultEngineProxyProfile() profile.DiscoveryService = noderec.ServiceLMStudio - b := brokerWithPrepositionedProfile(profile) + b := brokerWithEngineProxyProfile(profile) b.regCache = relay.NewRegistrationCache() b.engineProxy(profile).backendPort.Store(int32(backendPort)) updates := attachAdvertiserProxy(t, b, profile, proxyPort) - b.reconcileAdvertisePrepositioned(profile, backend.Client()) + b.reconcileAdvertiseEngine(profile, backend.Client()) registrations := b.regCache.Snapshot() if len(registrations) != 1 || registrations[0].Service != profile.DiscoveryService || registrations[0].Port != proxyPort { @@ -147,7 +147,7 @@ func TestPrepositionedAdvertiserTracksBackendHealth(t *testing.T) { } backend.Close() - b.reconcileAdvertisePrepositioned(profile, backend.Client()) + b.reconcileAdvertiseEngine(profile, backend.Client()) if got := b.regCache.Snapshot(); len(got) != 0 { t.Fatalf("unhealthy engine remained advertised: %+v", got) } @@ -156,10 +156,10 @@ func TestPrepositionedAdvertiserTracksBackendHealth(t *testing.T) { } } -func TestPrepositionedAdvertiserRejectsSelfForwardLoop(t *testing.T) { - profile := testPrepositionedProfile() +func TestEngineAdvertiserRejectsSelfForwardLoop(t *testing.T) { + profile := testDefaultEngineProxyProfile() profile.DiscoveryService = noderec.ServiceLMStudio - b := brokerWithPrepositionedProfile(profile) + b := brokerWithEngineProxyProfile(profile) b.regCache = relay.NewRegistrationCache() b.regCache.Register(noderec.RegisterParams{Service: profile.DiscoveryService, Port: 44000}) b.engineProxy(profile).backendPort.Store(44000) @@ -167,7 +167,7 @@ func TestPrepositionedAdvertiserRejectsSelfForwardLoop(t *testing.T) { // A nil client proves the collision check short-circuits before probing the // facade as though it were the backend. - b.reconcileAdvertisePrepositioned(profile, nil) + b.reconcileAdvertiseEngine(profile, nil) if got := b.regCache.Snapshot(); len(got) != 0 { t.Fatalf("self-forwarding facade remained advertised: %+v", got) } diff --git a/services/nvpair-ui-broker/broker.go b/services/nvpair-ui-broker/broker.go index aa83ec44..4e1dcacb 100644 --- a/services/nvpair-ui-broker/broker.go +++ b/services/nvpair-ui-broker/broker.go @@ -769,23 +769,20 @@ func (b *Broker) enableEngineFacadeWithPortCheck( alias ollamaHostAlias, available func(int) bool, ) error { - if profile.Ownership == prepositionedEngine { - return b.enableProxyFacadeWithFallback( - ctx, - pp, - b.prepositionedFacadeSpec(profile), - func(failed int) int { - return b.prepositionedFallbackPortWithCheck(profile, failed, available) - }, - ) - } switch profile.Name { case ollamaProxyProfile.Name: return b.enableProxyFacadeWithFallback(ctx, pp, b.ollamaFacadeSpec(alias), b.ollamaFallbackPort) case lmstudioProxyProfile.Name: return b.enableProxyFacadeWithFallback(ctx, pp, b.lmstudioFacadeSpec(), b.lmstudioFallbackPort) default: - return fmt.Errorf("no facade spec for engine %q", profile.Name) + return b.enableProxyFacadeWithFallback( + ctx, + pp, + b.defaultEngineFacadeSpec(profile), + func(failed int) int { + return b.defaultEngineFallbackPortWithCheck(profile, failed, available) + }, + ) } } @@ -799,9 +796,6 @@ func (b *Broker) enableEngineFacadeWithPortCheck( // alias. Finishing is what returns the alias, which is only correct once this // engine is known not to be coming up. func (b *Broker) blockAndFinishEngineProxy(profile engineProxyProfile) { - if profile.Ownership == prepositionedEngine { - return - } switch profile.Name { case ollamaProxyProfile.Name: if b.ollamaState().managedFacade.Load() { @@ -814,7 +808,7 @@ func (b *Broker) blockAndFinishEngineProxy(profile engineProxyProfile) { } b.finishLMStudioProxyTerminal() default: - slog.Warn("no terminal handling for engine", "engine", profile.Name) + // Other engines have no compatibility-port claim or startup gate. } } @@ -1009,8 +1003,8 @@ func (b *Broker) forwardProxyProcessNotification( } default: profile, known := engineProxyProfileFor(engine) - if known && profile.Ownership == prepositionedEngine { - b.forwardPrepositionedProxyNotification(profile, method, params) + if known { + b.forwardDefaultEngineProxyNotification(profile, method, params) return } slog.Warn("proxy addressed a notification without a handler", @@ -2234,12 +2228,15 @@ func (b *Broker) Serve(ctx context.Context) error { // the broker can resolve ownership. availabilityRunners := []func(context.Context){b.runAutoAdvertise, b.runAutoAdvertiseLMStudio} for _, profile := range engineProxyProfiles { - if profile.Ownership != prepositionedEngine || !b.proxyEnabled(profile) { + if !b.proxyEnabled(profile) { + continue + } + if profile.Name == ollamaProxyProfile.Name || profile.Name == lmstudioProxyProfile.Name { continue } profile := profile availabilityRunners = append(availabilityRunners, func(ctx context.Context) { - b.runAutoAdvertisePrepositioned(ctx, profile) + b.runAutoAdvertiseEngine(ctx, profile) }) } go b.runEngineAvailabilityAfterPortGates(ctx, availabilityRunners...) diff --git a/services/nvpair-ui-broker/prepositionedproxy_test.go b/services/nvpair-ui-broker/enginefacade_test.go similarity index 74% rename from services/nvpair-ui-broker/prepositionedproxy_test.go rename to services/nvpair-ui-broker/enginefacade_test.go index 46bf4ffe..93251056 100644 --- a/services/nvpair-ui-broker/prepositionedproxy_test.go +++ b/services/nvpair-ui-broker/enginefacade_test.go @@ -13,7 +13,7 @@ import ( "nvpair-shared/engines" ) -func testPrepositionedProfile() engineProxyProfile { +func testDefaultEngineProxyProfile() engineProxyProfile { return engineProxyProfile{ Engine: engines.Engine{ Name: "fixedtest", @@ -22,12 +22,12 @@ func testPrepositionedProfile() engineProxyProfile { EnginePortBase: 1234, PortFile: "fixedtest-proxy-port.json", }, - Ownership: prepositionedEngine, + Ownership: managedEngine, HealthProbePath: "/health", } } -func brokerWithPrepositionedProfile(profile engineProxyProfile) *Broker { +func brokerWithEngineProxyProfile(profile engineProxyProfile) *Broker { b := &Broker{} b.engineProxiesOnce.Do(func() { b.engineProxies = map[string]*engineProxyRuntime{ @@ -37,10 +37,10 @@ func brokerWithPrepositionedProfile(profile engineProxyProfile) *Broker { return b } -func TestPrepositionedFacadeRetriesAwayFromReservedPorts(t *testing.T) { +func TestDefaultEngineFacadeRetriesAwayFromReservedPorts(t *testing.T) { isolateOllamaHostTestConfig(t) - profile := testPrepositionedProfile() - b := brokerWithPrepositionedProfile(profile) + profile := testDefaultEngineProxyProfile() + b := brokerWithEngineProxyProfile(profile) proxyClient, proxyServer := net.Pipe() t.Cleanup(func() { @@ -60,7 +60,7 @@ func TestPrepositionedFacadeRetriesAwayFromReservedPorts(t *testing.T) { func(int) bool { return true }, ) if err != nil { - t.Fatalf("enable prepositioned facade: %v", err) + t.Fatalf("enable engine facade: %v", err) } first, second := <-attempts, <-attempts if first.Port != profile.FacadePort { @@ -74,23 +74,16 @@ func TestPrepositionedFacadeRetriesAwayFromReservedPorts(t *testing.T) { t.Fatal("fallback retry could restore the port that just failed") } - restart := b.prepositionedFacadeSpec(profile) + restart := b.defaultEngineFacadeSpec(profile) if restart.Port != second.Port || !restart.IgnorePersistedPort { t.Fatalf("restart spec = %+v, want explicit fallback port %d", restart, second.Port) } } -func TestPrepositionedProfileNeverTakesBackendOwnership(t *testing.T) { - profile := testPrepositionedProfile() - if profile.mayMoveRunningEngine() { - t.Fatal("prepositioned strategy may move a running backend") - } - if profile.blocksOnOccupiedFacade() { - t.Fatal("prepositioned strategy blocks instead of moving only its facade") - } - - b := brokerWithPrepositionedProfile(profile) - b.preparePrepositionedFacade(profile) +func TestDefaultEngineFacadePreservesEnginePort(t *testing.T) { + profile := testDefaultEngineProxyProfile() + b := brokerWithEngineProxyProfile(profile) + b.prepareDefaultEngineFacade(profile) state := b.engineProxy(profile) if got := int(state.backendPort.Load()); got != profile.EnginePortBase { t.Fatalf("backend port = %d, want fixed base %d", got, profile.EnginePortBase) @@ -103,9 +96,8 @@ func TestPrepositionedProfileNeverTakesBackendOwnership(t *testing.T) { } } -func TestPrepositionedNotificationPreservesFacadeAddress(t *testing.T) { - profile := lmstudioProxyProfile - profile.Ownership = prepositionedEngine +func TestDefaultEngineNotificationPreservesFacadeAddress(t *testing.T) { + profile := mustEngineProxyProfile("llamacpp") client, server := net.Pipe() t.Cleanup(func() { _ = client.Close() @@ -116,10 +108,10 @@ func TestPrepositionedNotificationPreservesFacadeAddress(t *testing.T) { b.setEngineProxySubscribed(profile, true) b.proxyMu.Unlock() - payload := json.RawMessage(`{"port":1234}`) + payload := json.RawMessage(`{"port":8080}`) done := make(chan struct{}) go func() { - b.forwardPrepositionedProxyNotification(profile, profile.addressed("ready"), payload) + b.forwardDefaultEngineProxyNotification(profile, profile.addressed("ready"), payload) close(done) }() if err := client.SetReadDeadline(time.Now().Add(2 * time.Second)); err != nil { diff --git a/services/nvpair-ui-broker/engineproxy.go b/services/nvpair-ui-broker/engineproxy.go index 975882f7..43a5ff91 100644 --- a/services/nvpair-ui-broker/engineproxy.go +++ b/services/nvpair-ui-broker/engineproxy.go @@ -9,11 +9,9 @@ package main // the broker about it. What this file adds is the one thing only the broker // needs: whether it may reposition the engine's process while it is running. // -// That single question decides every place the two engines' port choreography -// diverges, which is why it is a named enum rather than a set of booleans or a -// bag of function pointers. A hook would only move the divergent bodies into -// this file; naming the reason keeps them where they belong and makes the -// difference reviewable. +// Ownership governs relocation authority, not facade setup. Engines normally +// keep their configured port while the broker places their facade. Ollama and +// LM Studio additionally need engine-specific compatibility-port takeover. import ( "context" @@ -30,7 +28,7 @@ import ( "nvpair-shared/noderec" ) -// engineOwnership selects the broker's port-ownership strategy. +// engineOwnership describes the broker's authority to request engine relocation. type engineOwnership int const ( @@ -40,15 +38,11 @@ const ( // it holds the facade port. Ollama. adoptedEngine engineOwnership = iota - // managedEngine — engine-manager launched it in identified command mode - // and has an official stop command for it, so it may be stopped and - // repositioned before the proxy starts. LM Studio. + // managedEngine — engine-manager can stop and reposition a process it + // owns, or an identified command-mode runtime with an official stop + // command. It still refuses unknown or unowned processes. LM Studio and + // llama.cpp; only LM Studio needs automatic compatibility-port takeover. managedEngine - - // prepositionedEngine — engine-manager already owns the process and keeps - // it on EnginePortBase. The broker only places the facade, never gates - // engine requests or takes over/repositions the backend. - prepositionedEngine ) // engineProxyProfile is everything the broker needs to supervise one engine's @@ -56,9 +50,8 @@ const ( type engineProxyProfile struct { engines.Engine - // Ownership decides the port choreography: plan-then-commit for an adopted - // engine, move-then-verify for a managed one, or facade-only placement for - // a prepositioned one. It is the only judgment call in adding an engine. + // Ownership determines whether a running engine may be relocated when an + // engine-specific compatibility-port takeover requires it. Ownership engineOwnership // HealthProbePath is the path whose 200 means "this engine is answering". @@ -168,10 +161,10 @@ var engineProxyProfiles = buildEngineProxyProfiles() func buildEngineProxyProfiles() []engineProxyProfile { brokerOnly := map[string]engineProxyProfile{ "ollama": {Ownership: adoptedEngine, HealthProbePath: "/"}, - // LM Studio is the one engine engine-manager may move while running: - // its identified command-mode runtime has an official stop command. + // LM Studio's command-mode runtime has an official stop command; + // llama.cpp's managed process is stopped directly by engine-manager. "lmstudio": {Ownership: managedEngine, HealthProbePath: "/v1/models"}, - "llamacpp": {Ownership: prepositionedEngine, HealthProbePath: "/health"}, + "llamacpp": {Ownership: managedEngine, HealthProbePath: "/health"}, } out := make([]engineProxyProfile, 0, len(engines.All())) for _, e := range engines.All() { @@ -300,9 +293,9 @@ func (b *Broker) enableProxyFacadeWithFallback( return b.enableProxyFacade(parent, p, spec) } -// prepositionedFacadeSpec prefers the stock facade port, while preserving a +// defaultEngineFacadeSpec prefers the stock facade port, while preserving a // fallback or explicit port already selected for this broker lifetime. -func (b *Broker) prepositionedFacadeSpec(profile engineProxyProfile) enableFacadeRequest { +func (b *Broker) defaultEngineFacadeSpec(profile engineProxyProfile) enableFacadeRequest { spec := enableFacadeRequest{Engine: profile.Name, Port: profile.FacadePort} if port := int(b.engineProxy(profile).startupPort.Load()); port != 0 { spec.Port = port @@ -311,9 +304,9 @@ func (b *Broker) prepositionedFacadeSpec(profile engineProxyProfile) enableFacad return spec } -// prepositionedFallbackPortWithCheck keeps a fallback off the fixed backend +// defaultEngineFallbackPortWithCheck keeps a fallback off the engine's default // port and every sibling's facade/backend/persisted ports. -func (b *Broker) prepositionedFallbackPortWithCheck( +func (b *Broker) defaultEngineFallbackPortWithCheck( profile engineProxyProfile, failed int, available func(int) bool, ) int { excluded := []int{failed, profile.EnginePortBase} @@ -396,11 +389,11 @@ func (b *Broker) proxyEnabled(p engineProxyProfile) bool { // prepareEnabledFacades prepares port ownership for the engines the broker is // actually going to front, in table order — Ollama first, because its alias -// reservation constrains later engines. A prepositioned backend needs no gate -// or move; recording its fixed backend port is its entire preparation. +// reservation constrains later engines. The default path records the engine +// port without a gate or automatic relocation. // // The enablement check belongs here and not downstream, because preparation is -// not read-only: for a managed engine the backend move runs inside it, so +// not read-only: LM Studio's engine move runs inside it, so // preparing an engine whose proxy is never started relocates that engine off // its own stock port and leaves nothing serving it. Ollama cannot show the // symptom, since its move is deferred until its proxy proves it holds the @@ -410,20 +403,18 @@ func (b *Broker) prepareEnabledFacades() { if !b.proxyEnabled(profile) { continue } - switch { - case profile.Name == ollamaProxyProfile.Name: + switch profile.Name { + case ollamaProxyProfile.Name: b.prepareManagedOllamaFacade() - case profile.Name == lmstudioProxyProfile.Name: + case lmstudioProxyProfile.Name: b.prepareManagedLMStudioFacade() - case profile.Ownership == prepositionedEngine: - b.preparePrepositionedFacade(profile) default: - slog.Warn("no port preparation strategy for engine", "engine", profile.Name) + b.prepareDefaultEngineFacade(profile) } } } -func (b *Broker) preparePrepositionedFacade(profile engineProxyProfile) { +func (b *Broker) prepareDefaultEngineFacade(profile engineProxyProfile) { if b.prepareExplicitEngineSettings(profile.Name) { return } @@ -517,10 +508,9 @@ func (b *Broker) forwardEngineProxyNotification(profile engineProxyProfile, meth } } -// forwardPrepositionedProxyNotification is the facade-only notification path: -// there is no ownership or readiness reconciliation because the backend stays -// fixed on EnginePortBase. -func (b *Broker) forwardPrepositionedProxyNotification(profile engineProxyProfile, method string, params json.RawMessage) { +// forwardDefaultEngineProxyNotification relays a facade notification without +// engine-specific compatibility-port reconciliation. +func (b *Broker) forwardDefaultEngineProxyNotification(profile engineProxyProfile, method string, params json.RawMessage) { method, addressed := facadeMethodFor(profile, method) if !addressed { return diff --git a/services/nvpair-ui-broker/engineproxy_test.go b/services/nvpair-ui-broker/engineproxy_test.go index 418668f9..cac620c8 100644 --- a/services/nvpair-ui-broker/engineproxy_test.go +++ b/services/nvpair-ui-broker/engineproxy_test.go @@ -93,8 +93,8 @@ func TestBrokerConstantsMatchTheEngineTable(t *testing.T) { } } -// Ownership is the one judgment call in adding an engine, so every value in -// the table today are pinned explicitly. Getting these backwards does not fail +// Every engine's relocation authority is pinned explicitly. Getting these +// backwards does not fail // to compile — it silently changes which engine the broker believes it may stop. func TestEngineOwnershipAssignments(t *testing.T) { for _, tc := range []struct { @@ -103,7 +103,7 @@ func TestEngineOwnershipAssignments(t *testing.T) { }{ {"ollama", adoptedEngine}, {"lmstudio", managedEngine}, - {"llamacpp", prepositionedEngine}, + {"llamacpp", managedEngine}, } { p, ok := engineProxyProfileFor(tc.engine) if !ok { diff --git a/services/nvpair-ui-broker/enginesettings_review_test.go b/services/nvpair-ui-broker/enginesettings_review_test.go index 33418945..41b66d8b 100644 --- a/services/nvpair-ui-broker/enginesettings_review_test.go +++ b/services/nvpair-ui-broker/enginesettings_review_test.go @@ -64,7 +64,7 @@ func TestExplicitSettingsBindFailurePreservesChosenPort(t *testing.T) { case "lmstudio": b.forwardLMStudioProxyNotification("error", failure) default: - b.forwardPrepositionedProxyNotification(profile, profile.addressed("error"), failure) + b.forwardDefaultEngineProxyNotification(profile, profile.addressed("error"), failure) } if got := b.engineProxy(profile).startupPort.Load(); got != requested { t.Fatalf("bind notification changed chosen port to %d", got) diff --git a/services/nvpair-ui-broker/proxyport.go b/services/nvpair-ui-broker/proxyport.go index 01edd5f1..298a3c02 100644 --- a/services/nvpair-ui-broker/proxyport.go +++ b/services/nvpair-ui-broker/proxyport.go @@ -177,16 +177,13 @@ func (b *Broker) configureProxySupervisorCallbacks(sup *supervisor) { // than looping over a single helper: its gate re-checks whether a backend move // is pending, so clearing that move state is what actually reopens the path. func (b *Broker) finishEngineProxyStartup(profile engineProxyProfile) { - if profile.Ownership == prepositionedEngine { - return - } switch profile.Name { case ollamaProxyProfile.Name: b.finishOllamaProxyTerminal() case lmstudioProxyProfile.Name: b.finishLMStudioProxyTerminal() default: - slog.Warn("no startup-gate finisher for engine", "engine", profile.Name) + // Other engines have no compatibility-port startup gate. } } From a6746f57f664e70f2bdcc10b07642b70f6355b8e Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Mon, 5 Oct 2026 11:57:54 -0700 Subject: [PATCH 65/70] test(broker): preserve llama.cpp behavior under managed ownership Signed-off-by: Sherief Farouk --- .../nvpair-ui-broker/enginefacade_test.go | 145 ++++++++++++++++-- 1 file changed, 132 insertions(+), 13 deletions(-) diff --git a/services/nvpair-ui-broker/enginefacade_test.go b/services/nvpair-ui-broker/enginefacade_test.go index 93251056..9ccb01b8 100644 --- a/services/nvpair-ui-broker/enginefacade_test.go +++ b/services/nvpair-ui-broker/enginefacade_test.go @@ -7,10 +7,12 @@ import ( "context" "encoding/json" "net" + "sync/atomic" "testing" "time" "nvpair-shared/engines" + settings "nvpair-shared/enginesettings" ) func testDefaultEngineProxyProfile() engineProxyProfile { @@ -80,23 +82,133 @@ func TestDefaultEngineFacadeRetriesAwayFromReservedPorts(t *testing.T) { } } -func TestDefaultEngineFacadePreservesEnginePort(t *testing.T) { - profile := testDefaultEngineProxyProfile() - b := brokerWithEngineProxyProfile(profile) - b.prepareDefaultEngineFacade(profile) - state := b.engineProxy(profile) - if got := int(state.backendPort.Load()); got != profile.EnginePortBase { - t.Fatalf("backend port = %d, want fixed base %d", got, profile.EnginePortBase) +func TestLlamaCPPFacadePreparationPreservesConfiguredPorts(t *testing.T) { + profile := mustEngineProxyProfile("llamacpp") + for _, tc := range []struct { + name string + serverPort int + proxyPort int + explicit bool + }{ + {name: "manifest default", serverPort: profile.EnginePortBase}, + {name: "explicit settings", serverPort: 18081, proxyPort: 18080, explicit: true}, + } { + t.Run(tc.name, func(t *testing.T) { + b := &Broker{ + proxyPath: "test-proxy", + proxyEngines: []string{profile.Name}, + engineSettingsLoaded: true, + } + if tc.explicit { + b.engineSettings = map[string]*engineSettingsRecord{ + profile.Name: {Explicit: true, Snapshot: settings.Snapshot{ + Settings: settings.Config{ServerPort: tc.serverPort, ProxyPort: tc.proxyPort}, + }}, + } + } + worker, codec := newTestRPCWorkerPipe(t) + b.setEngineMgr(worker) + var calls atomic.Int32 + go func() { + for { + msg, err := codec.Read() + if err != nil { + return + } + calls.Add(1) + if err := codec.Respond(msg.ID, ollamaPortStatus{Running: true, Port: tc.serverPort}); err != nil { + t.Errorf("respond to unexpected engine request %s: %v", msg.Method, err) + return + } + } + }() + + b.prepareEnabledFacades() + + if got := calls.Load(); got != 0 { + t.Fatalf("preparation issued %d engine requests, want no probing or relocation", got) + } + state := b.engineProxy(profile) + if got := int(state.backendPort.Load()); got != tc.serverPort { + t.Fatalf("engine port = %d, want %d", got, tc.serverPort) + } + if got := int(state.startupPort.Load()); got != tc.proxyPort { + t.Fatalf("startup proxy port = %d, want %d", got, tc.proxyPort) + } + if got := state.explicitSettings.Load(); got != tc.explicit { + t.Fatalf("explicit settings = %v, want %v", got, tc.explicit) + } + }) } +} + +func TestLlamaCPPProxyTerminalHandlingPreservesEngineState(t *testing.T) { + profile := mustEngineProxyProfile("llamacpp") + b := &Broker{ollamaPortReady: make(chan struct{}), lmstudioPortReady: make(chan struct{})} + state := b.engineProxy(profile) + state.backendPort.Store(int32(profile.EnginePortBase)) + state.startupPort.Store(18080) state.managedFacade.Store(true) + b.blockAndFinishEngineProxy(profile) b.finishEngineProxyStartup(profile) - if !state.managedFacade.Load() { - t.Fatal("ungated terminal handling mutated ownership state") + + if got := int(state.backendPort.Load()); got != profile.EnginePortBase { + t.Fatalf("engine port = %d, want %d", got, profile.EnginePortBase) + } + if got := state.startupPort.Load(); got != 18080 || !state.managedFacade.Load() { + t.Fatalf("terminal handling changed facade state: port=%d managed=%v", got, state.managedFacade.Load()) + } + for _, gate := range []struct { + name string + ready <-chan struct{} + }{{"Ollama", b.ollamaPortReady}, {"LM Studio", b.lmstudioPortReady}} { + select { + case <-gate.ready: + t.Fatalf("llama.cpp terminal handling released %s's gate", gate.name) + default: + } + } +} + +func TestLlamaCPPEngineStatusRelaysBeforeOtherPortGates(t *testing.T) { + profile := mustEngineProxyProfile("llamacpp") + b := brokerWithEngineStatus(t, profile.Name, profile.EnginePortBase) + b.ollamaPortReady = make(chan struct{}) + b.lmstudioPortReady = make(chan struct{}) + b.managedOllamaBackend.Store(managedOllamaBackendStart) + client, server := net.Pipe() + t.Cleanup(func() { + close(b.ollamaPortReady) + close(b.lmstudioPortReady) + _ = client.Close() + _ = server.Close() + }) + b.codec = NewCodec(server) + id := json.RawMessage(`1`) + + b.relayToEngine(&Message{ID: &id, Method: "engine:status", Params: json.RawMessage(`{"engine":"llamacpp"}`)}) + + if err := client.SetReadDeadline(time.Now().Add(2 * time.Second)); err != nil { + t.Fatalf("set response deadline: %v", err) + } + response, err := NewCodec(client).Read() + if err != nil { + t.Fatalf("read llama.cpp status while other port gates are pending: %v", err) + } + if response.Error != nil { + t.Fatalf("llama.cpp status failed: %+v", response.Error) + } + var status ollamaPortStatus + if err := json.Unmarshal(response.Result, &status); err != nil { + t.Fatalf("decode llama.cpp status: %v", err) + } + if status.Port != profile.EnginePortBase { + t.Fatalf("status port = %d, want %d", status.Port, profile.EnginePortBase) } } -func TestDefaultEngineNotificationPreservesFacadeAddress(t *testing.T) { +func TestLlamaCPPProxyNotificationDispatchPreservesFacadeAddress(t *testing.T) { profile := mustEngineProxyProfile("llamacpp") client, server := net.Pipe() t.Cleanup(func() { @@ -111,7 +223,7 @@ func TestDefaultEngineNotificationPreservesFacadeAddress(t *testing.T) { payload := json.RawMessage(`{"port":8080}`) done := make(chan struct{}) go func() { - b.forwardDefaultEngineProxyNotification(profile, profile.addressed("ready"), payload) + b.forwardProxyProcessNotification(0, 0, profile.addressed("ready"), payload) close(done) }() if err := client.SetReadDeadline(time.Now().Add(2 * time.Second)); err != nil { @@ -121,8 +233,15 @@ func TestDefaultEngineNotificationPreservesFacadeAddress(t *testing.T) { if err != nil { t.Fatalf("read forwarded notification: %v", err) } - if msg.Method != profile.ComponentName()+":ready" || string(msg.Params) != string(payload) { - t.Fatalf("forwarded notification = %s %s", msg.Method, msg.Params) + if msg.Method != profile.ComponentName()+":ready" { + t.Fatalf("notification method = %s, want %s:ready", msg.Method, profile.ComponentName()) + } + var ready proxyReadyParams + if err := json.Unmarshal(msg.Params, &ready); err != nil { + t.Fatalf("decode ready notification: %v", err) + } + if ready.Port != 8080 { + t.Fatalf("ready port = %d, want 8080", ready.Port) } select { case <-done: From 176e57a699c524254156a1912af68a94a24974cd Mon Sep 17 00:00:00 2001 From: Sherief Farouk Date: Mon, 5 Oct 2026 11:57:54 -0700 Subject: [PATCH 66/70] docs(broker): clarify managed llama.cpp port behavior Signed-off-by: Sherief Farouk --- services/nvpair-ui-broker/README.md | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/services/nvpair-ui-broker/README.md b/services/nvpair-ui-broker/README.md index b0833fe9..c875b6b6 100644 --- a/services/nvpair-ui-broker/README.md +++ b/services/nvpair-ui-broker/README.md @@ -122,9 +122,13 @@ withdrawn or answered with an error on its own, leaving the others serving. Ollama's standalone default is `:11435`; with managed port ownership enabled (the default), the broker starts settings and engine-manager first, claims `:11434` with that facade, and only then moves a stopped default-port Ollama backend to a free port. Custom backend ports are preserved. When the inherited `OLLAMA_HOST` names a distinct local plaintext port, the broker also gives the facade that normalized loopback-only alias so clients already using the variable enter the same routing path; `localhost` reserves both canonical loopback families atomically, while remote and HTTPS targets are ignored. The alias port is reserved against every configured engine, local or remote engine start override, every facade's control plane, and the managed Ollama and LM Studio backend port plans, so a backend that has to move can never land on the alias. A running Ollama or unknown owner on either requested port is never stopped or moved: the primary uses a safe fallback when needed, and an occupied alias remains with its owner while the broker reports a warning. -llama.cpp is prepositioned by its manifest on `:8081`; the broker never takes -over or moves that process. Its default-enabled facade uses `:8080` or a safe -fallback without colliding with the fixed backend port. +llama.cpp is a managed engine: engine-manager owns its lifecycle and starts it +on its configured loopback port, `:8081` by default. The broker places its +default-enabled facade on `:8080` or a safe fallback, without automatically +relocating the engine or adding a compatibility-port ownership gate. Fallback +selection excludes the default engine port and other engines' reserved ports. +Explicit engine and proxy settings are preserved; a bind failure on an explicitly +chosen proxy port is reported instead of selecting a fallback. For automatic model-bearing inference, every facade combines scheduler pending counts and GPU pressure with the process-wide reservation map under one lock before forwarding, so concurrent requests distribute without an artificial delay or a round trip through the scheduler. A reservation is released when its request ends and moves with a failover, so a node stops counting as loaded as soon as it stops working. Manual pins, model-owner tiers, and the complete failover list keep their existing precedence. The broker otherwise treats the proxy as **optional and non-fatal**. From c004de4cd4d0e7e52443b85a1662b5833e1596cc Mon Sep 17 00:00:00 2001 From: Terve Date: Mon, 5 Oct 2026 16:35:17 -0400 Subject: [PATCH 67/70] Name the app NVIDIA PAIR in the VC++ runtime failure dialogs The Visual C++ Runtime install failures were the only user-visible installer messages still saying "Personal AI Router". The standalone services installer now uses PRODUCT_DISPLAY_NAME, which its own comment reserves for user-visible text, rather than PRODUCT_NAME, which has to stay exact because the uninstall registry key is keyed on it. Signed-off-by: Terve --- desktop/scripts/build/installer.nsh | 4 ++-- services/installer/nvpair-setup.nsi | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/desktop/scripts/build/installer.nsh b/desktop/scripts/build/installer.nsh index 2c2668b2..19165a43 100644 --- a/desktop/scripts/build/installer.nsh +++ b/desktop/scripts/build/installer.nsh @@ -288,7 +288,7 @@ ${if} ${Errors} ClearErrors Delete "$INSTDIR\resources\installer-tools\VC_redist.x64.exe" - MessageBox MB_OK|MB_ICONSTOP "Personal AI Router could not start the Microsoft Visual C++ Runtime installer.$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK + MessageBox MB_OK|MB_ICONSTOP "NVIDIA PAIR could not start the Microsoft Visual C++ Runtime installer.$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK SetErrorLevel 3 Quit ${endif} @@ -305,7 +305,7 @@ DetailPrint "Microsoft Visual C++ Runtime installed; Windows restart required." SetRebootFlag true ${else} - MessageBox MB_OK|MB_ICONSTOP "Personal AI Router could not install the Microsoft Visual C++ Runtime (exit code $8).$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK + MessageBox MB_OK|MB_ICONSTOP "NVIDIA PAIR could not install the Microsoft Visual C++ Runtime (exit code $8).$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK SetErrorLevel 3 Quit ${endif} diff --git a/services/installer/nvpair-setup.nsi b/services/installer/nvpair-setup.nsi index 7a300e8f..8bb6c8c1 100644 --- a/services/installer/nvpair-setup.nsi +++ b/services/installer/nvpair-setup.nsi @@ -141,7 +141,7 @@ FunctionEnd ${if} ${Errors} ClearErrors Delete "$PLUGINSDIR\VC_redist.x64.exe" - MessageBox MB_OK|MB_ICONSTOP "${PRODUCT_NAME} could not start the Microsoft Visual C++ Runtime installer.$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK + MessageBox MB_OK|MB_ICONSTOP "${PRODUCT_DISPLAY_NAME} could not start the Microsoft Visual C++ Runtime installer.$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK SetErrorLevel 3 Quit ${endif} @@ -158,7 +158,7 @@ FunctionEnd DetailPrint "Microsoft Visual C++ Runtime installed; Windows restart required." SetRebootFlag true ${else} - MessageBox MB_OK|MB_ICONSTOP "${PRODUCT_NAME} could not install the Microsoft Visual C++ Runtime (exit code $R4).$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK + MessageBox MB_OK|MB_ICONSTOP "${PRODUCT_DISPLAY_NAME} could not install the Microsoft Visual C++ Runtime (exit code $R4).$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK SetErrorLevel 3 Quit ${endif} From 1fc0ba5742ad156ec81ed970c9f0ecb39fc136d3 Mon Sep 17 00:00:00 2001 From: Terve Date: Tue, 6 Oct 2026 15:38:49 -0400 Subject: [PATCH 68/70] Fix llama.cpp install on macOS and Linux Installing llama.cpp failed with `engine "llamacpp" was not detected after install` on every macOS and Linux machine. The download verified and the archive extracted; the manifest was looking in the wrong place. detect and runtime.bin named {install_dir}/build/bin/llama-server, which is where a local cmake build puts its output, not where the published release archives unpack. Every llama.cpp release tarball wraps its contents in a directory named for the build tag, so the binary landed at {install_dir}/llama-b11146/llama-server and nothing matched. Each tar extraction now strips that wrapper and the paths name {install_dir}/llama-server, which converges macOS and Linux on the flat layout Windows already had -- its zip is not wrapped, which is why Windows was the one platform that worked. Linux strips both archives, whose wrapper directories are named differently, and its LD_LIBRARY_PATH follows the binary. Stripping rather than naming the wrapper keeps the release tag in one place. Spelling it in detect would mean editing four blocks in lockstep with every URL bump, which is the shape of mistake this already was. TestBundledInstallLayout downloads each archive, extracts it the way the manifest says to, and checks the detect path appears -- the comparison nobody was making. It is gated on NVPAIR_LIVE_LAYOUT because the archives are gigabytes, by environment variable rather than the `live` build tag, which does not currently compile in this package. TestBundledRuntimeBinIsDetected adds the part that needs no download: runtime.bin has to be one of the detect paths, so an engine cannot report installed and then launch something else. It asserts nothing about how deep a path may be -- llama.cpp wraps and Ollama's Linux archive is already bin/ and lib/, so there is no shared convention to pin. Signed-off-by: Terve --- .../install_linux_test.go | 12 +- .../installlayout_test.go | 163 ++++++++++++++++++ services/nvpair-engine-manager/launch_test.go | 2 +- .../manifests/llamacpp.json | 24 +-- 4 files changed, 185 insertions(+), 16 deletions(-) create mode 100644 services/nvpair-engine-manager/installlayout_test.go diff --git a/services/nvpair-engine-manager/install_linux_test.go b/services/nvpair-engine-manager/install_linux_test.go index 3b167cbd..b4a88ac5 100644 --- a/services/nvpair-engine-manager/install_linux_test.go +++ b/services/nvpair-engine-manager/install_linux_test.go @@ -19,8 +19,14 @@ import ( ) func TestLlamaCPPInstallPreservesShellArtifactArguments(t *testing.T) { - serverArchive := testTarGZIP(t, "build/bin/llama-server", "server") - cudartArchive := testTarGZIP(t, "build/bin/libcudart.so.12", "runtime") + // Both archives wrap their contents in one top-level directory, as every + // llama.cpp release tarball does, and the two wrappers are named + // differently — the server's after the build tag, the CUDA runtime's after + // the whole artifact. The install strips one component from each so the + // binaries and their libraries land side by side in the install directory, + // which is where detect and runtime.bin look. + serverArchive := testTarGZIP(t, "llama-b11146/llama-server", "server") + cudartArchive := testTarGZIP(t, "cudart-llama-b11146-bin-ubuntu-cuda-12.8-x64/libcudart.so.12", "runtime") archives := map[string][]byte{ "/server.tar.gz": serverArchive, "/cudart.tar.gz": cudartArchive, @@ -65,7 +71,7 @@ func TestLlamaCPPInstallPreservesShellArtifactArguments(t *testing.T) { "llama-server": "server", "libcudart.so.12": "runtime", } { - data, err := os.ReadFile(filepath.Join(baseDir, "llamacpp", "build", "bin", name)) + data, err := os.ReadFile(filepath.Join(baseDir, "llamacpp", name)) if err != nil { t.Fatalf("read extracted %s: %v", name, err) } diff --git a/services/nvpair-engine-manager/installlayout_test.go b/services/nvpair-engine-manager/installlayout_test.go new file mode 100644 index 00000000..7b539b2b --- /dev/null +++ b/services/nvpair-engine-manager/installlayout_test.go @@ -0,0 +1,163 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +package main + +import ( + "os" + "os/exec" + "path/filepath" + "runtime" + "strings" + "testing" +) + +// bundledManifestSet loads the compiled-in manifests the way main.go does. +func bundledManifestSet(t *testing.T) map[string]*Manifest { + t.Helper() + reg := NewRegistry() + if err := reg.LoadFS(bundledManifests, "manifests"); err != nil { + t.Fatalf("load bundled manifests: %v", err) + } + out := map[string]*Manifest{} + for _, name := range reg.Names() { + manifest, ok := reg.Get(name) + if !ok { + t.Fatalf("registry lost manifest %q", name) + } + out[name] = manifest + } + if len(out) == 0 { + t.Fatal("no bundled manifests loaded") + } + return out +} + +// TestBundledRuntimeBinIsDetected keeps detection and launch pointing at the +// same file. +// +// A detect path is a claim about what the install produces, and nothing checks +// it at install time: the runner extracts, looks for the declared path, finds +// nothing, and reports "was not detected after install" with no way to say +// why. That is what llama.cpp did on macOS and Linux, where the manifest named +// the layout of a local cmake build (build/bin/llama-server) rather than of the +// published release archive. +// +// Whether a path matches the archive is only knowable from the archive, so the +// real check is TestLiveBundledInstallLayout, which downloads each one. What is +// checkable here is that the two paths agree: detecting one file and launching +// another lets an engine report installed and then fail to start. +// +// There is deliberately no rule about how deep a detect path may be. Archive +// shapes are the vendor's choice and they differ — llama.cpp wraps everything +// in a build-tagged directory the install has to strip, while Ollama's Linux +// archive is already bin/ and lib/ and must not be stripped. A convention +// asserted here would only encode one vendor's habit as if it were a contract. +func TestBundledRuntimeBinIsDetected(t *testing.T) { + for name, manifest := range bundledManifestSet(t) { + for key, platform := range manifest.Platforms { + bin := platform.Runtime.Bin + if bin == "" || len(platform.Detect) == 0 { + continue + } + found := false + for _, candidate := range platform.Detect { + if candidate == bin { + found = true + break + } + } + if !found { + t.Errorf("%s/%s: runtime.bin %q is not among the detect paths %v — the engine would report installed and then fail to start", + name, key, bin, platform.Detect) + } + } + } +} + +// TestBundledInstallLayout downloads every archive the bundled manifests +// install from, extracts it the way the manifest says to, and checks the detect +// path appears. This is the check that was missing when llama.cpp shipped a +// detect path no release archive could satisfy. +// +// Gated on an environment variable rather than a build tag: the archives run to +// gigabytes, but the `live` tag does not currently compile in this package, and +// a verification nobody can run is not one. +// +// Extraction runs through the host's tar, which libarchive-backed tar handles +// for .tar.gz, .tar.zst and .zip alike, so one host can verify another +// platform's archive. The --strip-components the manifest declares is applied, +// because that flag is what decides whether the detect path resolves. A +// platform whose install is not a tar invocation (Windows llama.cpp uses +// Expand-Archive) is still covered: only the extraction mechanism differs, and +// the archive it reads is the one fetched here. +// +// NVPAIR_LIVE_LAYOUT=1 go test -run TestBundledInstallLayout -v -timeout 3600s +func TestBundledInstallLayout(t *testing.T) { + if os.Getenv("NVPAIR_LIVE_LAYOUT") == "" { + t.Skip("set NVPAIR_LIVE_LAYOUT=1 to download every engine archive and verify its layout") + } + for name, manifest := range bundledManifestSet(t) { + for key, platform := range manifest.Platforms { + urls := archiveURLs(platform) + if len(urls) == 0 { + continue // vendor script or detect-only engine; nothing to unpack + } + t.Run(name+"/"+key, func(t *testing.T) { + installDir := t.TempDir() + strip := strings.Contains(strings.Join(platform.Install.Run, " "), "--strip-components=1") + for _, url := range urls { + extractArchive(t, url, installDir, strip) + } + for _, candidate := range platform.Detect { + if !strings.HasPrefix(candidate, "{install_dir}") { + continue + } + relative := strings.TrimLeft(strings.TrimPrefix(candidate, "{install_dir}"), `/\`) + // Manifests spell Windows paths with backslashes; the host + // separator is what the extracted tree uses. + relative = filepath.FromSlash(strings.ReplaceAll(relative, `\`, "/")) + if _, err := os.Stat(filepath.Join(installDir, relative)); err != nil { + t.Errorf("detect path %q is absent after extracting %v (strip=%v): %v", + candidate, urls, strip, err) + } + } + }) + } + } +} + +// archiveURLs lists the downloads a platform's install unpacks, in the order +// the manifest extracts them. +func archiveURLs(platform Platform) []string { + if platform.Install == nil || len(platform.Install.Run) == 0 { + return nil + } + var urls []string + if platform.Install.Fetch != nil { + urls = append(urls, platform.Install.Fetch.URL) + } + for _, artifact := range platform.Install.Artifacts { + urls = append(urls, artifact.URL) + } + return urls +} + +func extractArchive(t *testing.T, url, installDir string, strip bool) { + t.Helper() + archive := filepath.Join(t.TempDir(), filepath.Base(url)) + // curl rather than net/http: these are large, redirected downloads and the + // point here is the archive's interior, not the transfer. + fetch := exec.Command("curl", "-sSL", "--fail", "--max-time", "900", "-o", archive, url) + if out, err := fetch.CombinedOutput(); err != nil { + t.Skipf("cannot download %s (%v): %s", url, err, strings.TrimSpace(string(out))) + } + args := []string{"-xf", archive, "-C", installDir} + if strip { + args = append(args, "--strip-components=1") + } + extract := exec.Command("tar", args...) + if out, err := extract.CombinedOutput(); err != nil { + t.Fatalf("tar %v on %s failed (%v): %s", args, runtime.GOOS, err, strings.TrimSpace(string(out))) + } +} diff --git a/services/nvpair-engine-manager/launch_test.go b/services/nvpair-engine-manager/launch_test.go index 3f3f4815..ac8e9607 100644 --- a/services/nvpair-engine-manager/launch_test.go +++ b/services/nvpair-engine-manager/launch_test.go @@ -51,7 +51,7 @@ func TestResolvedLaunchMatchesBundledEngines(t *testing.T) { } want = []string{"LLAMA_CACHE=/test path-models", bin, "--sleep-idle-seconds", "300", "--host", "127.0.0.1", "--port", "12345", "--cors-origins", ""} if strings.HasPrefix(platformKey, "linux/") { - want = append([]string{"LD_LIBRARY_PATH=/test path/build/bin"}, want...) + want = append([]string{"LD_LIBRARY_PATH=/test path"}, want...) } } if err != nil { diff --git a/services/nvpair-engine-manager/manifests/llamacpp.json b/services/nvpair-engine-manager/manifests/llamacpp.json index a0e95a2a..c22b2b0f 100644 --- a/services/nvpair-engine-manager/manifests/llamacpp.json +++ b/services/nvpair-engine-manager/manifests/llamacpp.json @@ -47,46 +47,46 @@ "runtime": { "bin": "{install_dir}\\llama-server.exe" } }, "linux/amd64": { - "detect": ["{install_dir}/build/bin/llama-server"], + "detect": ["{install_dir}/llama-server"], "install": { "artifacts": [ { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-ubuntu-cuda-12.8-x64.tar.gz", "sha256": "c2ab9e19838513ff69d1af8d999ad717dd3c7ee4714ac04c7ed5ab9077c50e4e" }, { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-b11146-bin-ubuntu-cuda-12.8-x64.tar.gz", "sha256": "1466daea60aad1144819e151b2bae19d54556cf1da6c129c4f55a5ded2637c25" } ], - "run": ["sh", "-c", "tar -xzf \"$1\" -C \"$3\" && tar -xzf \"$2\" -C \"$3\"", "llamacpp-install", "{download_server}", "{download_cudart}", "{install_dir}"] + "run": ["sh", "-c", "tar -xzf \"$1\" -C \"$3\" --strip-components=1 && tar -xzf \"$2\" -C \"$3\" --strip-components=1", "llamacpp-install", "{download_server}", "{download_cudart}", "{install_dir}"] }, "uninstall": { "run": ["rm", "-rf", "{install_dir}"] }, - "runtime": { "bin": "{install_dir}/build/bin/llama-server", "env": { "LD_LIBRARY_PATH": "{install_dir}/build/bin" } } + "runtime": { "bin": "{install_dir}/llama-server", "env": { "LD_LIBRARY_PATH": "{install_dir}" } } }, "linux/arm64": { - "detect": ["{install_dir}/build/bin/llama-server"], + "detect": ["{install_dir}/llama-server"], "install": { "artifacts": [ { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-ubuntu-cuda-13.4-arm64.tar.gz", "sha256": "4e00496ab6cdee9c00afb11de3cb9d10f9da7e17147d8ed14ca3af05209b400f" }, { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-b11146-bin-ubuntu-cuda-13.4-arm64.tar.gz", "sha256": "7f46057efcba6338c58ed9f91c06f268c290f3a228fdbdd4bea229dd60b0e094" } ], - "run": ["sh", "-c", "tar -xzf \"$1\" -C \"$3\" && tar -xzf \"$2\" -C \"$3\"", "llamacpp-install", "{download_server}", "{download_cudart}", "{install_dir}"] + "run": ["sh", "-c", "tar -xzf \"$1\" -C \"$3\" --strip-components=1 && tar -xzf \"$2\" -C \"$3\" --strip-components=1", "llamacpp-install", "{download_server}", "{download_cudart}", "{install_dir}"] }, "uninstall": { "run": ["rm", "-rf", "{install_dir}"] }, - "runtime": { "bin": "{install_dir}/build/bin/llama-server", "env": { "LD_LIBRARY_PATH": "{install_dir}/build/bin" } } + "runtime": { "bin": "{install_dir}/llama-server", "env": { "LD_LIBRARY_PATH": "{install_dir}" } } }, "darwin/arm64": { - "detect": ["{install_dir}/build/bin/llama-server"], + "detect": ["{install_dir}/llama-server"], "install": { "fetch": { "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-macos-arm64.tar.gz", "sha256": "1ad3f9eff80edb9dbef4259ad564d1720612ef7eea48fa4afed0e54f5f3d5711" }, - "run": ["tar", "-xzf", "{download}", "-C", "{install_dir}"] + "run": ["tar", "-xzf", "{download}", "-C", "{install_dir}", "--strip-components=1"] }, "uninstall": { "run": ["rm", "-rf", "{install_dir}"] }, - "runtime": { "bin": "{install_dir}/build/bin/llama-server" } + "runtime": { "bin": "{install_dir}/llama-server" } }, "darwin/amd64": { - "detect": ["{install_dir}/build/bin/llama-server"], + "detect": ["{install_dir}/llama-server"], "install": { "fetch": { "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-macos-x64.tar.gz", "sha256": "305f0e3a17d2c01eb205cd0a62128357f1ec3b55329cb084d94e5ec0115d7a3b" }, - "run": ["tar", "-xzf", "{download}", "-C", "{install_dir}"] + "run": ["tar", "-xzf", "{download}", "-C", "{install_dir}", "--strip-components=1"] }, "uninstall": { "run": ["rm", "-rf", "{install_dir}"] }, - "runtime": { "bin": "{install_dir}/build/bin/llama-server" } + "runtime": { "bin": "{install_dir}/llama-server" } } }, "actions": { From a23398926cdc8eacde6c91142ee205edfbfaeaf7 Mon Sep 17 00:00:00 2001 From: Terve Date: Mon, 5 Oct 2026 17:18:39 -0400 Subject: [PATCH 69/70] Read the redistributable's Authenticode signature without PowerShell `npm run stage:vc-redist` asked PowerShell for the staged Visual C++ Redistributable's signature, so it only ran on Windows. The public `build:electron:win:x64` and `:arm64` scripts cross-build the Windows installers from Linux and macOS, where the staging step could not run at all, leaving the packaging precondition unsatisfiable off Windows. Read the signature out of the PE instead on those platforms: the certificate table's PKCS#7 SignedData, the image digest it covers, the embedded chain, and the version resource. Staging now fails unless the digest computed over these bytes matches the digest Microsoft signed, and unless the signing certificate chains by issuance and signature to a pinned Microsoft certificate authority. Windows still defers to the operating system, and both paths meet one policy in `validateAuthenticodeMetadata()`, so the recorded provenance is identical whichever host staged the package. Certificate validity windows, revocation, and the timestamp countersignature are deliberately not checked; a signing certificate that has expired since it signed is normal. Signed-off-by: Terve --- desktop/package.json | 2 +- scripts/stage-vc-redist.mjs | 25 +- scripts/windows-pe-signature.mjs | 460 ++++++++++++++++++++++++++ scripts/windows-pe-signature.test.mjs | 197 +++++++++++ 4 files changed, 678 insertions(+), 6 deletions(-) create mode 100644 scripts/windows-pe-signature.mjs create mode 100644 scripts/windows-pe-signature.test.mjs diff --git a/desktop/package.json b/desktop/package.json index 82229c06..30aa9d22 100644 --- a/desktop/package.json +++ b/desktop/package.json @@ -23,7 +23,7 @@ "typecheck:test": "tsc --noEmit -p tsconfig.test.json", "typecheck": "npm run typecheck:node && npm run typecheck:web && npm run typecheck:test", "test": "npm run test:unit", - "test:vc-redist": "node --test ../scripts/stage-vc-redist.test.mjs ../scripts/windows-vc-runtime-installers.test.mjs", + "test:vc-redist": "node --test ../scripts/stage-vc-redist.test.mjs ../scripts/windows-pe-signature.test.mjs ../scripts/windows-vc-runtime-installers.test.mjs", "test:unit": "npm run test:vc-redist && vitest run --project unit --passWithNoTests", "test:unit:coverage": "vitest run --project unit --coverage --passWithNoTests && tsx scripts/coverage-summary.ts", "test:unit:watch": "vitest --project unit", diff --git a/scripts/stage-vc-redist.mjs b/scripts/stage-vc-redist.mjs index a4f0ea15..b2245cef 100644 --- a/scripts/stage-vc-redist.mjs +++ b/scripts/stage-vc-redist.mjs @@ -10,6 +10,13 @@ * Authenticode establishes the publisher and integrity, while the minimum * version rejects a validly signed rollback. The observed version and SHA-256 * are recorded for release provenance rather than pinned as build inputs. + * + * On Windows the operating system reads the signature. Everywhere else it is + * read out of the PE directly, because `build:electron:win:*` cross-builds the + * Windows installers from Linux and macOS, where there is no PowerShell to ask. + * Both paths report the same fields and meet the same policy below; see + * scripts/windows-pe-signature.mjs for what the direct read does and does not + * establish. */ import { spawnSync } from 'node:child_process' @@ -18,6 +25,8 @@ import { mkdirSync, readFileSync, renameSync, rmSync, writeFileSync } from 'node import { dirname, join, resolve } from 'node:path' import { fileURLToPath, pathToFileURL } from 'node:url' +import { readPeSignatureMetadata } from './windows-pe-signature.mjs' + export const VC_REDIST_SOURCE_URL = 'https://aka.ms/vc14/vc_redist.x64.exe' export const VC_REDIST_MINIMUM_VERSION = '14.51.36247.0' @@ -69,7 +78,7 @@ function isMicrosoftSigner(subject) { export function validateAuthenticodeMetadata(metadata) { if (typeof metadata !== 'object' || metadata === null || Array.isArray(metadata)) { - throw new Error('PowerShell returned invalid Authenticode metadata.') + throw new Error('The Authenticode probe returned invalid metadata.') } if (metadata.status !== 'Valid') { throw new Error( @@ -103,7 +112,7 @@ export function validateAuthenticodeMetadata(metadata) { } } -function inspectAuthenticode(filePath) { +function windowsAuthenticodeMetadata(filePath) { const command = [ "$ErrorActionPreference = 'Stop'", '$signature = Get-AuthenticodeSignature -LiteralPath $env:NVPAIR_VC_REDIST_PATH', @@ -137,14 +146,20 @@ function inspectAuthenticode(filePath) { ) } - let metadata try { - metadata = JSON.parse(result.stdout.trim()) + return JSON.parse(result.stdout.trim()) } catch (error) { const message = error instanceof Error ? error.message : String(error) throw new Error(`Unable to parse Authenticode metadata: ${message}`) } - return validateAuthenticodeMetadata(metadata) +} + +function inspectAuthenticode(filePath) { + return validateAuthenticodeMetadata( + process.platform === 'win32' + ? windowsAuthenticodeMetadata(filePath) + : readPeSignatureMetadata(filePath) + ) } function sha256(filePath) { diff --git a/scripts/windows-pe-signature.mjs b/scripts/windows-pe-signature.mjs new file mode 100644 index 00000000..b3a4582b --- /dev/null +++ b/scripts/windows-pe-signature.mjs @@ -0,0 +1,460 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +/** + * Read an Authenticode signature out of a Windows PE file on any platform. + * + * The Windows installers cross-build on Linux, where `Get-AuthenticodeSignature` + * does not exist, so the staged Visual C++ Redistributable has to be inspected + * without the operating system's help. This module reads the PE's certificate + * table directly and reports the same fields the PowerShell probe does. + * + * What it establishes: the file carries a PKCS#7 signature; the digest that was + * signed is the digest of these bytes; the signing certificate chains, by + * issuance and by signature, to the Microsoft certificate authority pinned + * below; and the version the PE reports about itself. What it does not + * establish: certificate validity windows, revocation, or the RFC 3161 + * countersignature. A signing certificate that has expired since it signed is + * normal and still accepted here, which is the main reason this is not a + * general-purpose Authenticode verifier — on Windows the operating system + * remains the one making that judgement. + */ + +import { X509Certificate, createHash } from 'node:crypto' +import { readFileSync } from 'node:fs' + +// Authenticode stores a PKCS#7 SignedData whose encapsulated content is an +// SpcIndirectDataContent carrying the digest of the image. +const SIGNED_DATA_OID = '1.2.840.113549.1.7.2' +const SPC_INDIRECT_DATA_OID = '1.3.6.1.4.1.311.2.1.4' + +const DIGEST_ALGORITHM_OIDS = new Map([ + ['1.3.14.3.2.26', 'sha1'], + ['2.16.840.1.101.3.4.2.1', 'sha256'], + ['2.16.840.1.101.3.4.2.2', 'sha384'], + ['2.16.840.1.101.3.4.2.3', 'sha512'] +]) + +// The certificate authority the embedded chain is required to terminate in. +// +// A PE embeds the signing certificate and its issuers but not the root, so this +// pins the issuing CA rather than a root. Pinning is what makes the signer +// subject meaningful: without an anchor, any self-signed certificate could name +// itself Microsoft Corporation. When Microsoft moves to a new CA the staging +// fails with the new fingerprint in the message, and that fingerprint is added +// here after being confirmed against a Microsoft-published certificate. +export const MICROSOFT_ISSUER_FINGERPRINTS = new Set([ + // CN=Microsoft Code Signing PCA 2024, O=Microsoft Corporation, C=US — the + // last certificate in the chain embedded in VC_redist.x64.exe 14.51.36247.0. + '3D:AD:FA:F8:12:DD:1B:BA:EF:45:83:4C:CB:D1:88:F3:CD:97:13:9E:2E:D1:AC:A6:9C:2D:D6:30:82:14:2F:8F' +]) + +const PE32_MAGIC = 0x10b +const PE32PLUS_MAGIC = 0x20b +// Offsets inside the optional header. The checksum sits at the same place in +// both variants; the data directory array does not. +const CHECKSUM_OFFSET_IN_OPTIONAL_HEADER = 64 +const DATA_DIRECTORY_OFFSET_IN_OPTIONAL_HEADER = new Map([ + [PE32_MAGIC, 96], + [PE32PLUS_MAGIC, 112] +]) +const DATA_DIRECTORY_ENTRY_SIZE = 8 +const RESOURCE_DIRECTORY_INDEX = 2 +const CERTIFICATE_TABLE_INDEX = 4 + +const WIN_CERTIFICATE_HEADER_SIZE = 8 +const WIN_CERTIFICATE_TYPE_PKCS_SIGNED_DATA = 0x0002 + +const RESOURCE_TYPE_VERSION = 16 +const VS_FIXEDFILEINFO_SIGNATURE = 0xfeef04bd + +export function readTypeLengthValue(buffer, offset) { + if (offset + 2 > buffer.length) { + throw new Error('The signature is truncated: a DER header runs past the end.') + } + const tag = buffer[offset] + const firstLengthByte = buffer[offset + 1] + let length = firstLengthByte + let headerLength = 2 + + if ((firstLengthByte & 0x80) !== 0) { + const lengthByteCount = firstLengthByte & 0x7f + // Indefinite length is not legal in DER, and four bytes already covers + // any signature that fits the download size limit. + if (lengthByteCount === 0 || lengthByteCount > 4) { + throw new Error( + `The signature uses an unsupported DER length of ${lengthByteCount} bytes.` + ) + } + length = 0 + for (let index = 0; index < lengthByteCount; index += 1) { + length = length * 256 + buffer[offset + 2 + index] + } + headerLength = 2 + lengthByteCount + } + + const start = offset + headerLength + const end = start + length + if (end > buffer.length) { + throw new Error('The signature is truncated: a DER value runs past the end.') + } + // `offset` is the element itself, `start` only its content: a certificate + // has to be handed on with its own header intact. + return { tag, offset, start, end } +} + +export function derChildren(buffer, node) { + const nodes = [] + let offset = node.start + while (offset < node.end) { + const child = readTypeLengthValue(buffer, offset) + nodes.push(child) + offset = child.end + } + return nodes +} + +function expectTag(node, tag, description) { + if (node.tag !== tag) { + throw new Error( + `The signature is malformed: expected ${description} (tag 0x${tag.toString(16)}), ` + + `found tag 0x${node.tag.toString(16)}.` + ) + } + return node +} + +export function decodeObjectIdentifier(buffer, node) { + const bytes = buffer.subarray(node.start, node.end) + if (bytes.length === 0) { + throw new Error('The signature is malformed: an empty object identifier.') + } + const parts = [Math.floor(bytes[0] / 40), bytes[0] % 40] + let value = 0 + for (let index = 1; index < bytes.length; index += 1) { + value = value * 128 + (bytes[index] & 0x7f) + if ((bytes[index] & 0x80) === 0) { + parts.push(value) + value = 0 + } + } + return parts.join('.') +} + +/** + * The PE offsets this module needs: what to exclude from the image digest, and + * where the certificate table and resources live. + */ +function readPortableExecutableLayout(buffer) { + if (buffer.length < 64 || buffer[0] !== 0x4d || buffer[1] !== 0x5a) { + throw new Error('The file is not a Windows executable (no MZ header).') + } + const peHeaderOffset = buffer.readUInt32LE(0x3c) + if (peHeaderOffset + 24 > buffer.length || buffer.readUInt32LE(peHeaderOffset) !== 0x00004550) { + throw new Error('The file is not a Windows executable (no PE signature).') + } + + const sectionCount = buffer.readUInt16LE(peHeaderOffset + 6) + const optionalHeaderSize = buffer.readUInt16LE(peHeaderOffset + 20) + const optionalHeaderOffset = peHeaderOffset + 24 + const magic = buffer.readUInt16LE(optionalHeaderOffset) + const dataDirectoryOffset = DATA_DIRECTORY_OFFSET_IN_OPTIONAL_HEADER.get(magic) + if (dataDirectoryOffset === undefined) { + throw new Error( + `The executable has an unknown optional header magic 0x${magic.toString(16)}.` + ) + } + + const directoryOffset = index => + optionalHeaderOffset + dataDirectoryOffset + index * DATA_DIRECTORY_ENTRY_SIZE + const readDirectory = index => ({ + address: buffer.readUInt32LE(directoryOffset(index)), + size: buffer.readUInt32LE(directoryOffset(index) + 4) + }) + + const sectionTableOffset = optionalHeaderOffset + optionalHeaderSize + const sections = [] + for (let index = 0; index < sectionCount; index += 1) { + const offset = sectionTableOffset + index * 40 + sections.push({ + virtualAddress: buffer.readUInt32LE(offset + 12), + virtualSize: buffer.readUInt32LE(offset + 8), + rawOffset: buffer.readUInt32LE(offset + 20), + rawSize: buffer.readUInt32LE(offset + 16) + }) + } + + return { + checksumOffset: optionalHeaderOffset + CHECKSUM_OFFSET_IN_OPTIONAL_HEADER, + certificateDirectoryOffset: directoryOffset(CERTIFICATE_TABLE_INDEX), + certificateTable: readDirectory(CERTIFICATE_TABLE_INDEX), + resourceDirectory: readDirectory(RESOURCE_DIRECTORY_INDEX), + sections + } +} + +function fileOffsetForAddress(layout, address) { + for (const section of layout.sections) { + const end = section.virtualAddress + Math.max(section.virtualSize, section.rawSize) + if (address >= section.virtualAddress && address < end) { + return section.rawOffset + (address - section.virtualAddress) + } + } + throw new Error(`The executable has no section containing address 0x${address.toString(16)}.`) +} + +/** + * The Authenticode image digest: the whole file except the three regions a + * signature cannot cover — its own checksum, the certificate table's data + * directory entry, and the certificate table itself. + */ +function imageDigest(buffer, layout, algorithm) { + const certificateTableOffset = layout.certificateTable.address + const hash = createHash(algorithm) + hash.update(buffer.subarray(0, layout.checksumOffset)) + hash.update(buffer.subarray(layout.checksumOffset + 4, layout.certificateDirectoryOffset)) + hash.update( + buffer.subarray( + layout.certificateDirectoryOffset + DATA_DIRECTORY_ENTRY_SIZE, + certificateTableOffset + ) + ) + // Anything appended after the certificate table is covered, so a trailing + // payload cannot be swapped out. For these packages there is none. + hash.update(buffer.subarray(certificateTableOffset + layout.certificateTable.size)) + return hash.digest('hex') +} + +function readCertificateTable(buffer, layout) { + const { address, size } = layout.certificateTable + if (address === 0 || size === 0) { + throw new Error('The executable is not signed: it has no certificate table.') + } + if (address + size > buffer.length) { + throw new Error('The executable declares a certificate table past the end of the file.') + } + const certificateType = buffer.readUInt16LE(address + 6) + if (certificateType !== WIN_CERTIFICATE_TYPE_PKCS_SIGNED_DATA) { + throw new Error( + `The executable's certificate table holds type ${certificateType}, not PKCS#7 signed data.` + ) + } + const declaredLength = buffer.readUInt32LE(address) + return buffer.subarray(address + WIN_CERTIFICATE_HEADER_SIZE, address + declaredLength) +} + +/** + * Pull the certificates and the signed image digest out of the PKCS#7 blob. + */ +function readSignedData(pkcs7) { + const contentInfo = expectTag(readTypeLengthValue(pkcs7, 0), 0x30, 'a PKCS#7 ContentInfo') + const [contentType, wrappedContent] = derChildren(pkcs7, contentInfo) + if ( + decodeObjectIdentifier(pkcs7, expectTag(contentType, 0x06, 'a content type')) !== + SIGNED_DATA_OID + ) { + throw new Error('The executable is not signed with PKCS#7 SignedData.') + } + + const signedData = expectTag( + derChildren(pkcs7, expectTag(wrappedContent, 0xa0, 'the SignedData wrapper'))[0], + 0x30, + 'a SignedData' + ) + const members = derChildren(pkcs7, signedData) + // version, digestAlgorithms, contentInfo, then the optional [0] certificates + // and [1] CRLs, then signerInfos. + const encapsulated = expectTag(members[2], 0x30, 'an encapsulated ContentInfo') + const certificateSet = members.find(member => member.tag === 0xa0) + if (certificateSet === undefined) { + throw new Error('The signature carries no certificates.') + } + + const [encapsulatedType, encapsulatedContent] = derChildren(pkcs7, encapsulated) + if ( + decodeObjectIdentifier(pkcs7, expectTag(encapsulatedType, 0x06, 'a content type')) !== + SPC_INDIRECT_DATA_OID + ) { + throw new Error('The signature does not carry an Authenticode image digest.') + } + + const indirectData = expectTag( + derChildren(pkcs7, expectTag(encapsulatedContent, 0xa0, 'the content wrapper'))[0], + 0x30, + 'an SpcIndirectDataContent' + ) + const digestInfo = expectTag(derChildren(pkcs7, indirectData)[1], 0x30, 'the signed digest') + const [algorithmIdentifier, digestValue] = derChildren(pkcs7, digestInfo) + const algorithmOid = decodeObjectIdentifier( + pkcs7, + expectTag( + derChildren(pkcs7, expectTag(algorithmIdentifier, 0x30, 'a digest algorithm'))[0], + 0x06, + 'a digest algorithm identifier' + ) + ) + const algorithm = DIGEST_ALGORITHM_OIDS.get(algorithmOid) + if (algorithm === undefined) { + throw new Error(`The signature uses an unsupported digest algorithm ${algorithmOid}.`) + } + + return { + algorithm, + digest: pkcs7 + .subarray( + expectTag(digestValue, 0x04, 'the signed digest value').start, + digestValue.end + ) + .toString('hex'), + certificates: derChildren(pkcs7, certificateSet).map( + node => new X509Certificate(pkcs7.subarray(node.offset, node.end)) + ) + } +} + +/** + * Walk the embedded certificates from the signing certificate outwards, and + * require the chain to terminate in a pinned Microsoft certificate authority. + * + * Only issuance and signature are checked, not the validity window: Microsoft's + * signing certificates outlive their own signatures by design, and the + * countersignature recording when the signing happened is not verified here. + */ +export function verifyChainToMicrosoftIssuer(certificates) { + const issuesAnother = certificate => + certificates.some(other => other !== certificate && other.checkIssued(certificate)) + const leaf = certificates.find(certificate => !issuesAnother(certificate)) + if (leaf === undefined) { + throw new Error( + 'The signature has no signing certificate: every certificate issues another.' + ) + } + + let current = leaf + const seen = new Set([current.fingerprint256]) + for (;;) { + const issuer = certificates.find( + candidate => candidate !== current && current.checkIssued(candidate) + ) + if (issuer === undefined) break + if (!current.verify(issuer.publicKey)) { + throw new Error( + `The certificate chain is broken: "${distinguishedName(current.subject)}" is not ` + + 'signed by its issuer.' + ) + } + if (seen.has(issuer.fingerprint256)) { + throw new Error('The certificate chain loops back on itself.') + } + seen.add(issuer.fingerprint256) + current = issuer + } + + if (!MICROSOFT_ISSUER_FINGERPRINTS.has(current.fingerprint256)) { + throw new Error( + `The certificate chain ends at "${distinguishedName(current.subject)}" ` + + `(${current.fingerprint256}), which is not a pinned Microsoft ` + + 'certificate authority.' + ) + } + return leaf +} + +/** + * One level of the resource tree: named entries first, then the ones addressed + * by numeric id. A directory entry's offset is relative to the tree's root. + */ +function resourceEntries(buffer, directoryOffset) { + const namedCount = buffer.readUInt16LE(directoryOffset + 12) + const idCount = buffer.readUInt16LE(directoryOffset + 14) + const entries = [] + for (let index = 0; index < namedCount + idCount; index += 1) { + const offset = directoryOffset + 16 + index * 8 + const name = buffer.readUInt32LE(offset) + const data = buffer.readUInt32LE(offset + 4) + entries.push({ + id: (name & 0x80000000) === 0 ? name : null, + isDirectory: (data & 0x80000000) !== 0, + offset: data & 0x7fffffff + }) + } + return entries +} + +/** + * VS_FIXEDFILEINFO sits behind a UTF-16 key and alignment padding, so it is + * found by its signature rather than by counting bytes to it. + */ +function fixedFileInfoOffset(buffer, versionOffset) { + for (let offset = versionOffset; offset < versionOffset + 64; offset += 4) { + if (buffer.readUInt32LE(offset) === VS_FIXEDFILEINFO_SIGNATURE) return offset + } + throw new Error("The executable's version resource carries no VS_FIXEDFILEINFO.") +} + +/** + * The version the PE reports about itself, from its first RT_VERSION resource. + */ +function readVersionInfo(buffer, layout) { + const resourceRoot = fileOffsetForAddress(layout, layout.resourceDirectory.address) + const typeEntry = resourceEntries(buffer, resourceRoot).find( + entry => entry.id === RESOURCE_TYPE_VERSION && entry.isDirectory + ) + if (typeEntry === undefined) { + throw new Error('The executable carries no version resource.') + } + const nameEntry = resourceEntries(buffer, resourceRoot + typeEntry.offset)[0] + const languageEntry = resourceEntries(buffer, resourceRoot + nameEntry.offset)[0] + // The leaf addresses an IMAGE_RESOURCE_DATA_ENTRY, and that entry's own + // OffsetToData is an image address rather than a resource-relative one. + const dataEntryOffset = resourceRoot + languageEntry.offset + const versionOffset = fileOffsetForAddress(layout, buffer.readUInt32LE(dataEntryOffset)) + + const fixedInfoOffset = fixedFileInfoOffset(buffer, versionOffset) + const fileVersion = versionFromParts( + buffer.readUInt32LE(fixedInfoOffset + 8), + buffer.readUInt32LE(fixedInfoOffset + 12) + ) + const productVersion = versionFromParts( + buffer.readUInt32LE(fixedInfoOffset + 16), + buffer.readUInt32LE(fixedInfoOffset + 20) + ) + return { fileVersion, productVersion } +} + +export function versionFromParts(most, least) { + return [most >>> 16, most & 0xffff, least >>> 16, least & 0xffff].join('.') +} + +/** + * Node lists a subject one attribute per line, least significant first, and + * spells stateOrProvinceName `ST`. PowerShell prints a single comma-separated + * line, most significant first, and spells it `S`. Following PowerShell keeps + * the recorded signer identical whichever platform staged the package. + */ +export function distinguishedName(subject) { + return subject + .split('\n') + .filter(attribute => attribute.length > 0) + .reverse() + .map(attribute => attribute.replace(/^ST=/, 'S=')) + .join(', ') +} + +export function readPeSignatureMetadata(filePath) { + const buffer = readFileSync(filePath) + const layout = readPortableExecutableLayout(buffer) + const { algorithm, digest, certificates } = readSignedData(readCertificateTable(buffer, layout)) + + const leaf = verifyChainToMicrosoftIssuer(certificates) + const computed = imageDigest(buffer, layout, algorithm) + const { fileVersion, productVersion } = readVersionInfo(buffer, layout) + + return { + status: computed === digest ? 'Valid' : 'HashMismatch', + signerSubject: distinguishedName(leaf.subject), + signerThumbprint: leaf.fingerprint.replaceAll(':', ''), + fileVersion, + productVersion + } +} diff --git a/scripts/windows-pe-signature.test.mjs b/scripts/windows-pe-signature.test.mjs new file mode 100644 index 00000000..f27e4665 --- /dev/null +++ b/scripts/windows-pe-signature.test.mjs @@ -0,0 +1,197 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +import assert from 'node:assert/strict' +import test from 'node:test' + +import { + decodeObjectIdentifier, + derChildren, + distinguishedName, + MICROSOFT_ISSUER_FINGERPRINTS, + readTypeLengthValue, + verifyChainToMicrosoftIssuer, + versionFromParts +} from './windows-pe-signature.mjs' + +const DER_SEQUENCE = 0x30 +const DER_OBJECT_IDENTIFIER = 0x06 + +const objectIdentifier = bytes => Buffer.from([DER_OBJECT_IDENTIFIER, bytes.length, ...bytes]) + +/** + * A certificate stood up from a name and the name of its issuer, exposing only + * what the chain walk uses. `checkIssued(candidate)` answers whether this + * certificate was issued by the candidate, the same direction Node's + * X509Certificate uses. + */ +function certificate({ name, fingerprint256, issuedBy = null, signatureValid = true }) { + return { + subject: `CN=${name}`, + fingerprint256, + publicKey: { name }, + checkIssued: candidate => issuedBy !== null && candidate.publicKey.name === issuedBy, + verify: publicKey => signatureValid && publicKey.name === issuedBy + } +} + +const PINNED_ISSUER_FINGERPRINT = [...MICROSOFT_ISSUER_FINGERPRINTS][0] + +test('every pinned issuer is recorded as a SHA-256 fingerprint', () => { + // The chain walk compares against `fingerprint256`, so a SHA-1 thumbprint + // pasted in here would reject every file instead of failing visibly. + assert.ok(MICROSOFT_ISSUER_FINGERPRINTS.size > 0) + for (const fingerprint of MICROSOFT_ISSUER_FINGERPRINTS) { + assert.match(fingerprint, /^([0-9A-F]{2}:){31}[0-9A-F]{2}$/) + } +}) + +test('readTypeLengthValue reads a short-form element', () => { + const buffer = Buffer.from([DER_SEQUENCE, 0x03, 0x01, 0x02, 0x03]) + assert.deepEqual(readTypeLengthValue(buffer, 0), { + tag: DER_SEQUENCE, + offset: 0, + start: 2, + end: 5 + }) +}) + +test('readTypeLengthValue keeps the element offset apart from its content', () => { + // A certificate has to be handed to X509Certificate with its own header, so + // `offset` addressing the element and `start` addressing the content is the + // distinction the signature reader depends on. + const buffer = Buffer.concat([ + Buffer.from([DER_SEQUENCE, 0x81, 0xc8]), + Buffer.alloc(0xc8, 0x41) + ]) + const node = readTypeLengthValue(buffer, 0) + assert.equal(node.offset, 0) + assert.equal(node.start, 3) + assert.equal(node.end, buffer.length) +}) + +test('readTypeLengthValue reads a two-byte length', () => { + const buffer = Buffer.concat([ + Buffer.from([DER_SEQUENCE, 0x82, 0x01, 0x00]), + Buffer.alloc(256, 0x41) + ]) + assert.equal(readTypeLengthValue(buffer, 0).end, 260) +}) + +test('readTypeLengthValue rejects a value that runs past the end', () => { + assert.throws( + () => readTypeLengthValue(Buffer.from([DER_SEQUENCE, 0x10, 0x00]), 0), + /truncated/ + ) +}) + +test('readTypeLengthValue rejects an unsupported length width', () => { + assert.throws( + () => readTypeLengthValue(Buffer.from([DER_SEQUENCE, 0x85, 0, 0, 0, 0, 0]), 0), + /unsupported DER length of 5 bytes/ + ) +}) + +test('derChildren walks the elements of a sequence', () => { + const buffer = Buffer.from([DER_SEQUENCE, 0x06, 0x02, 0x01, 0x07, 0x02, 0x02, 0x08, 0x09]) + const children = derChildren(buffer, readTypeLengthValue(buffer, 0)) + assert.deepEqual( + children.map(child => [child.tag, child.end - child.start]), + [ + [0x02, 1], + [0x02, 2] + ] + ) +}) + +test('decodeObjectIdentifier decodes the PKCS#7 and Authenticode identifiers', () => { + const signedData = objectIdentifier([0x2a, 0x86, 0x48, 0x86, 0xf7, 0x0d, 0x01, 0x07, 0x02]) + const spcIndirectData = objectIdentifier([ + 0x2b, 0x06, 0x01, 0x04, 0x01, 0x82, 0x37, 0x02, 0x01, 0x04 + ]) + assert.equal( + decodeObjectIdentifier(signedData, readTypeLengthValue(signedData, 0)), + '1.2.840.113549.1.7.2' + ) + // 311 needs two continuation bytes, so this covers the multi-byte arc. + assert.equal( + decodeObjectIdentifier(spcIndirectData, readTypeLengthValue(spcIndirectData, 0)), + '1.3.6.1.4.1.311.2.1.4' + ) +}) + +test('versionFromParts unpacks a VS_FIXEDFILEINFO version', () => { + // 14.51.36247.0, as the staged redistributable reports itself. + assert.equal(versionFromParts(0x000e0033, 0x8d970000), '14.51.36247.0') + assert.equal(versionFromParts(0, 0), '0.0.0.0') + assert.equal(versionFromParts(0xffffffff, 0xffffffff), '65535.65535.65535.65535') +}) + +test('distinguishedName reorders a subject the way PowerShell prints it', () => { + assert.equal( + distinguishedName('C=US\nST=Washington\nL=Redmond\nO=Microsoft Corporation\n'), + 'O=Microsoft Corporation, L=Redmond, S=Washington, C=US' + ) +}) + +test('verifyChainToMicrosoftIssuer returns the signing certificate', () => { + const issuer = certificate({ + name: 'Microsoft Code Signing PCA 2024', + fingerprint256: PINNED_ISSUER_FINGERPRINT + }) + const leaf = certificate({ + name: 'Microsoft Corporation', + fingerprint256: 'AA:BB', + issuedBy: 'Microsoft Code Signing PCA 2024' + }) + // Certificate order in the signature is not specified, so the walk has to + // find the signing certificate rather than take the first one. + assert.equal(verifyChainToMicrosoftIssuer([issuer, leaf]), leaf) +}) + +test('verifyChainToMicrosoftIssuer rejects a chain to an unpinned authority', () => { + const issuer = certificate({ name: 'Example Root CA', fingerprint256: 'CC:DD' }) + const leaf = certificate({ + name: 'Microsoft Corporation', + fingerprint256: 'AA:BB', + issuedBy: 'Example Root CA' + }) + assert.throws( + () => verifyChainToMicrosoftIssuer([leaf, issuer]), + /ends at "CN=Example Root CA" \(CC:DD\), which is not a pinned Microsoft/ + ) +}) + +test('verifyChainToMicrosoftIssuer rejects a certificate its issuer did not sign', () => { + const issuer = certificate({ + name: 'Microsoft Code Signing PCA 2024', + fingerprint256: PINNED_ISSUER_FINGERPRINT + }) + const leaf = certificate({ + name: 'Microsoft Corporation', + fingerprint256: 'AA:BB', + issuedBy: 'Microsoft Code Signing PCA 2024', + signatureValid: false + }) + assert.throws( + () => verifyChainToMicrosoftIssuer([leaf, issuer]), + /chain is broken: "CN=Microsoft Corporation" is not signed by its issuer/ + ) +}) + +test('verifyChainToMicrosoftIssuer rejects a chain that never terminates', () => { + const first = certificate({ name: 'First CA', fingerprint256: 'CC:DD', issuedBy: 'Second CA' }) + const second = certificate({ name: 'Second CA', fingerprint256: 'EE:FF', issuedBy: 'First CA' }) + const leaf = certificate({ + name: 'Microsoft Corporation', + fingerprint256: 'AA:BB', + issuedBy: 'First CA' + }) + assert.throws(() => verifyChainToMicrosoftIssuer([leaf, first, second]), /loops back on itself/) +}) + +test('verifyChainToMicrosoftIssuer rejects a signature with no signing certificate', () => { + const first = certificate({ name: 'First CA', fingerprint256: 'CC:DD', issuedBy: 'Second CA' }) + const second = certificate({ name: 'Second CA', fingerprint256: 'EE:FF', issuedBy: 'First CA' }) + assert.throws(() => verifyChainToMicrosoftIssuer([first, second]), /no signing certificate/) +}) From 0fa98666ea3ed1cd75c47ea6d3222fd308cafcb9 Mon Sep 17 00:00:00 2001 From: Terve Date: Wed, 7 Oct 2026 14:36:36 -0400 Subject: [PATCH 70/70] Stage the VC++ runtime from a PowerShell 7 shell on Windows `stage:vc-redist` failed on both Windows runners with "The 'Get-AuthenticodeSignature' command was found in the module 'Microsoft.PowerShell.Security', but the module could not be loaded". The build step runs under PowerShell 7, the windows-latest default. npm and node sit between it and the powershell.exe this script starts, so PowerShell 7 never strips its own entries from PSModulePath for the child, and Windows PowerShell inherits a module path that finds 7's Microsoft.PowerShell.Security first, which it cannot load (PowerShell/PowerShell#18530). Start Windows PowerShell without PSModulePath so it builds its own default. The same failure met anyone running build:win:* from a pwsh terminal, not only CI. Signed-off-by: Terve --- scripts/stage-vc-redist.mjs | 17 ++++++++++++++++- scripts/stage-vc-redist.test.mjs | 24 +++++++++++++++++++++++- 2 files changed, 39 insertions(+), 2 deletions(-) diff --git a/scripts/stage-vc-redist.mjs b/scripts/stage-vc-redist.mjs index b2245cef..d7729976 100644 --- a/scripts/stage-vc-redist.mjs +++ b/scripts/stage-vc-redist.mjs @@ -112,6 +112,21 @@ export function validateAuthenticodeMetadata(metadata) { } } +/** + * The environment for the Windows PowerShell child. A `PSModulePath` inherited + * from PowerShell 7 points Windows PowerShell at 7's incompatible + * Microsoft.PowerShell.Security, so `Get-AuthenticodeSignature` cannot load + * (PowerShell/PowerShell#18530). With the variable unset, Windows PowerShell + * builds its own default. Windows matches variable names case-insensitively. + */ +export function windowsPowerShellEnv(parentEnv, filePath) { + const env = Object.fromEntries( + Object.entries(parentEnv).filter(([name]) => name.toLowerCase() !== 'psmodulepath') + ) + env.NVPAIR_VC_REDIST_PATH = filePath + return env +} + function windowsAuthenticodeMetadata(filePath) { const command = [ "$ErrorActionPreference = 'Stop'", @@ -131,7 +146,7 @@ function windowsAuthenticodeMetadata(filePath) { ['-NoProfile', '-NonInteractive', '-Command', command], { encoding: 'utf8', - env: { ...process.env, NVPAIR_VC_REDIST_PATH: filePath }, + env: windowsPowerShellEnv(process.env, filePath), windowsHide: true } ) diff --git a/scripts/stage-vc-redist.test.mjs b/scripts/stage-vc-redist.test.mjs index 412cd8ec..5961c87a 100644 --- a/scripts/stage-vc-redist.test.mjs +++ b/scripts/stage-vc-redist.test.mjs @@ -8,7 +8,8 @@ import { compareVersions, createProvenance, validateAuthenticodeMetadata, - VC_REDIST_MINIMUM_VERSION + VC_REDIST_MINIMUM_VERSION, + windowsPowerShellEnv } from './stage-vc-redist.mjs' const MICROSOFT_SIGNATURE = { @@ -64,6 +65,27 @@ test('validateAuthenticodeMetadata rejects a signed rollback', () => { ) }) +test('windowsPowerShellEnv drops a PSModulePath inherited from PowerShell 7', () => { + const parentEnv = { + Path: 'C:\\Windows\\system32', + PSModulePath: 'C:\\Program Files\\PowerShell\\7\\Modules' + } + + assert.deepEqual(windowsPowerShellEnv(parentEnv, 'C:\\stage\\VC_redist.x64.exe'), { + Path: 'C:\\Windows\\system32', + NVPAIR_VC_REDIST_PATH: 'C:\\stage\\VC_redist.x64.exe' + }) + assert.equal(parentEnv.PSModulePath, 'C:\\Program Files\\PowerShell\\7\\Modules') +}) + +test('windowsPowerShellEnv drops PSModulePath whatever its case', () => { + const parentEnv = { PSMODULEPATH: 'C:\\Program Files\\PowerShell\\7\\Modules' } + + assert.deepEqual(windowsPowerShellEnv(parentEnv, 'C:\\stage\\VC_redist.x64.exe'), { + NVPAIR_VC_REDIST_PATH: 'C:\\stage\\VC_redist.x64.exe' + }) +}) + test('createProvenance records the verified package identity', () => { const resolvedUrl = 'https://download.visualstudio.microsoft.com/download/pr/package/VC_redist.x64.exe'