diff --git a/.cursor/rules/model-registry.mdc b/.cursor/rules/model-registry.mdc
index e66521a7..a7ee3dfa 100644
--- a/.cursor/rules/model-registry.mdc
+++ b/.cursor/rules/model-registry.mdc
@@ -1,5 +1,5 @@
---
-description: Electron-main engine model hub (Ollama committed list + LM Studio catalog)
+description: Electron-main engine model hub (locked Ollama + live Hugging Face catalogs)
alwaysApply: true
---
+
diff --git a/desktop/src/ui/components/EngineIcon.tsx b/desktop/src/ui/components/EngineIcon.tsx
index 4ee81367..bb1618a0 100644
--- a/desktop/src/ui/components/EngineIcon.tsx
+++ b/desktop/src/ui/components/EngineIcon.tsx
@@ -5,6 +5,7 @@ import type { CSSProperties } from 'react'
import { type EngineType } from '@/shared/types/engines'
import ollamaIcon from '@/ui/assets/engine-icons/ollama.png?inline'
import lmStudioIcon from '@/ui/assets/engine-icons/lm-studio.png?inline'
+import llamaCppIcon from '@/ui/assets/engine-icons/llama-cpp.svg?inline'
export default function EngineIcon({ type, size = 32 }: { type: EngineType; size?: number }) {
const dimension = `${size}px`
@@ -39,5 +40,13 @@ export default function EngineIcon({ type, size = 32 }: { type: EngineType; size
)
}
+ if (type === 'llama-cpp') {
+ return (
+
+
+
+ )
+ }
+
return null
}
diff --git a/desktop/src/ui/components/ModelHub/ModelHubContent.tsx b/desktop/src/ui/components/ModelHub/ModelHubContent.tsx
index 061b25f6..354709b7 100644
--- a/desktop/src/ui/components/ModelHub/ModelHubContent.tsx
+++ b/desktop/src/ui/components/ModelHub/ModelHubContent.tsx
@@ -8,7 +8,7 @@ import { ModelHubActions } from './ModelHubActions'
import { InlineErrorBanner } from '@/ui/components/InlineErrorBanner'
import { ModelEntry, SortState } from '@/ui/types/model-hub'
import { ModelHubList } from './ModelHubList'
-import { searchEngineHub } from '@/ui/utils/model-hub-search'
+import { mergeModelHubResults, searchEngineHub } from '@/ui/utils/model-hub-search'
import { resolveStoredSort, writeStoredSort } from '@/ui/utils/model-hub-content-storage'
import { isHubEntryDownloaded } from '@/ui/utils/match-downloaded-model'
import { EngineType } from '@/shared/types/engines'
@@ -42,7 +42,9 @@ export const ModelHubContent = ({
const [disabledListClick, setDisabledListClick] = useState(false)
const [loading, setLoading] = useState(true)
const [allModels, setAllModels] = useState([])
+ const [remoteModels, setRemoteModels] = useState([])
const [query, setQuery] = useState('')
+ const [searching, setSearching] = useState(false)
const [selectedModels, setSelectedModels] = useState>(new Set())
const [sort, setSort] = useState(() =>
engine ? resolveStoredSort(engine, SORT_DEFAULT_STATE) : { ...SORT_DEFAULT_STATE }
@@ -55,6 +57,8 @@ export const ModelHubContent = ({
// Monotonic generation guard so a stale in-flight fetch never overwrites
// results for the engine the user has since switched to.
const fetchGenRef = useRef(0)
+ const searchGenRef = useRef(0)
+ const submittedQueryRef = useRef('')
const loadModels = useCallback(async (backend: EngineType) => {
const cached = cacheRef.current.get(backend)
@@ -84,6 +88,9 @@ export const ModelHubContent = ({
}, [])
useEffect(() => {
+ setRemoteModels([])
+ setSearching(false)
+ submittedQueryRef.current = ''
if (!engine) {
setAllModels([])
setQuery('')
@@ -95,16 +102,18 @@ export const ModelHubContent = ({
setSort(resolveStoredSort(engine, SORT_DEFAULT_STATE))
void loadModels(engine)
return () => {
- // Invalidate any in-flight fetch for the previous engine.
+ // Invalidate in-flight population or search for the previous engine.
fetchGenRef.current += 1
+ searchGenRef.current += 1
}
}, [engine, loadModels])
const queryFiltered = useMemo(() => {
const q = query.toLowerCase().trim()
if (!q) return allModels
- return allModels.filter(m => m.name.toLowerCase().includes(q))
- }, [allModels, query])
+ const populatedMatches = allModels.filter(m => m.name.toLowerCase().includes(q))
+ return mergeModelHubResults(populatedMatches, remoteModels)
+ }, [allModels, query, remoteModels])
const visibleModels = useMemo(() => {
if (!engine || !downloadedModels || downloadedModels.length === 0) return queryFiltered
@@ -121,8 +130,8 @@ export const ModelHubContent = ({
[engine]
)
- const modelsRef = useRef(allModels)
- modelsRef.current = allModels
+ const modelsRef = useRef(queryFiltered)
+ modelsRef.current = queryFiltered
const handleSubmit = useCallback(
(ids: string[]) => {
@@ -162,7 +171,41 @@ export const ModelHubContent = ({
[multiple, onToggleSelectModel, handleSubmit]
)
- const handleSearchSubmit = useCallback((q: string) => setQuery(q), [])
+ const handleQueryChange = useCallback((nextQuery: string) => {
+ setQuery(nextQuery)
+ if (nextQuery.trim() === submittedQueryRef.current) return
+ searchGenRef.current += 1
+ submittedQueryRef.current = ''
+ setRemoteModels([])
+ setSearching(false)
+ }, [])
+
+ const handleSearch = useCallback(
+ async (nextQuery: string) => {
+ const normalized = nextQuery.trim()
+ setQuery(nextQuery)
+ if (engine !== 'llama-cpp' || normalized.length === 0) return
+
+ const generation = ++searchGenRef.current
+ submittedQueryRef.current = normalized
+ setRemoteModels([])
+ setSearching(true)
+ try {
+ const result = await searchEngineHub(engine, normalized)
+ if (searchGenRef.current !== generation) return
+ setRemoteModels(result)
+ } catch (error) {
+ if (searchGenRef.current !== generation) return
+ setErrors(prev => [
+ ...prev,
+ { id: performance.now().toString(), message: getErrorString(error) }
+ ])
+ } finally {
+ if (searchGenRef.current === generation) setSearching(false)
+ }
+ },
+ [engine]
+ )
return (
@@ -183,7 +226,10 @@ export const ModelHubContent = ({
)}
diff --git a/desktop/src/ui/components/ModelHub/ModelHubSearchBar.tsx b/desktop/src/ui/components/ModelHub/ModelHubSearchBar.tsx
index b531a309..4d5dcb28 100644
--- a/desktop/src/ui/components/ModelHub/ModelHubSearchBar.tsx
+++ b/desktop/src/ui/components/ModelHub/ModelHubSearchBar.tsx
@@ -1,36 +1,34 @@
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0
-import { useState } from 'react'
import { Button, Flex, TextInput } from '@nvidia/foundations-react-core'
import { Search } from '@/ui/components/icons'
import { SortState } from '@/ui/types/model-hub'
import { ModelHubFilterSort } from './ModelHubFilterSort'
interface ModelHubSearchBarProps {
- onSubmit: (query: string) => void
+ query: string
+ searching: boolean
+ onQueryChange: (query: string) => void
+ onSearch: (query: string) => void
sort: SortState
onSort: (sort: SortState) => void
onMenuOpenChange?: (open: boolean) => void
}
export const ModelHubSearchBar = ({
- onSubmit,
+ query,
+ searching,
+ onQueryChange,
+ onSearch,
sort,
onSort,
onMenuOpenChange
}: ModelHubSearchBarProps) => {
- const [query, setQuery] = useState('')
-
- const handleValueChange = (value: string) => {
- setQuery(value)
- // Results are cached in the renderer, so filter live as the user types.
- onSubmit(value)
- }
-
const handleKeyDown = (e: React.KeyboardEvent) => {
- if (e.key === 'Enter') {
- onSubmit(query)
+ if (e.key === 'Enter' && !searching && query.trim()) {
+ e.preventDefault()
+ onSearch(query)
}
}
@@ -41,17 +39,28 @@ export const ModelHubSearchBar = ({
placeholder="Search for a model"
value={query}
onKeyDown={handleKeyDown}
- onValueChange={handleValueChange}
+ onValueChange={onQueryChange}
className="flex-1 min-w-0 max-h-[32px]"
/>
)
diff --git a/desktop/src/ui/constants/engine-capabilities.ts b/desktop/src/ui/constants/engine-capabilities.ts
index 77e6f2bf..2626a2e4 100644
--- a/desktop/src/ui/constants/engine-capabilities.ts
+++ b/desktop/src/ui/constants/engine-capabilities.ts
@@ -43,5 +43,23 @@ export const EngineCapabilities: Record = {
// server. Deleting therefore interrupts inference and needs a warning.
restartsOnModelDelete: true,
engineHub: { label: 'LM Studio', url: 'https://lmstudio.ai/models' }
+ },
+ 'llama-cpp': {
+ hasExpiry: false,
+ hasEject: true,
+ hasInstall: ['win32', 'darwin', 'linux'],
+ hasEnginePort: true,
+ hasInstallPath: false,
+ hasProxyWebUI: false,
+ hasPreferredNode: false,
+ hasCrashAlert: false,
+ hasModelSearchOnlyWhenRunning: true,
+ modelOpsWhenStopped: false,
+ // The router removes cached models in-process via DELETE /models.
+ hasDeleteModel: true,
+ engineHub: {
+ label: 'llama.cpp',
+ url: 'https://huggingface.co/models?library=gguf'
+ }
}
}
diff --git a/desktop/src/ui/constants/welcome.ts b/desktop/src/ui/constants/welcome.ts
index 4f9fa04a..4db425f7 100644
--- a/desktop/src/ui/constants/welcome.ts
+++ b/desktop/src/ui/constants/welcome.ts
@@ -13,7 +13,8 @@ export const WELCOME_STEP_SUB_HEADINGS = ['', 'You can update later by clicking
export const WELCOME_ENGINE_DEFAULT_SELECTED: Record = {
ollama: true,
- 'lm-studio': true
+ 'lm-studio': true,
+ 'llama-cpp': true
}
export function getWelcomeEngineCandidates(os: PlatformDisplayName): EngineType[] {
diff --git a/desktop/src/ui/utils/match-downloaded-model.ts b/desktop/src/ui/utils/match-downloaded-model.ts
index 2edf8361..cdce8576 100644
--- a/desktop/src/ui/utils/match-downloaded-model.ts
+++ b/desktop/src/ui/utils/match-downloaded-model.ts
@@ -17,6 +17,8 @@ import type { ModelEntry } from '@/ui/types/model-hub'
* prefix so every quantization collapses to "downloaded".
* - LM Studio: the download is keyed by `pullKey = /` (or
* `//`); the hub id is `/`.
+ * - llama.cpp: router inventory and catalog use the same complete
+ * `/:` id, so only exact ids match.
*/
type DownloadedMatcher = (hubEntry: ModelEntry, downloaded: ModelItem) => boolean
@@ -36,9 +38,12 @@ const matchHfPullKeyOrName: DownloadedMatcher = (hubEntry, d) => {
return false
}
+const matchExactName: DownloadedMatcher = (hubEntry, downloaded) => downloaded.name === hubEntry.id
+
const MATCHERS: Partial> = {
ollama: matchOllama,
- 'lm-studio': matchHfPullKeyOrName
+ 'lm-studio': matchHfPullKeyOrName,
+ 'llama-cpp': matchExactName
}
export function isHubEntryDownloaded(
diff --git a/desktop/src/ui/utils/model-hub-search.ts b/desktop/src/ui/utils/model-hub-search.ts
index 8111cbec..e78a0ea6 100644
--- a/desktop/src/ui/utils/model-hub-search.ts
+++ b/desktop/src/ui/utils/model-hub-search.ts
@@ -38,13 +38,22 @@ function hubModelToEntry(m: EngineHubModel): ModelEntry {
}
}
+export function mergeModelHubResults(
+ populated: readonly ModelEntry[],
+ searched: readonly ModelEntry[]
+): ModelEntry[] {
+ const byId = new Map()
+ for (const model of populated) byId.set(model.id, model)
+ for (const model of searched) byId.set(model.id, model)
+ return Array.from(byId.values())
+}
+
/**
- * Fetch an engine's full model hub catalog from the service. The Electron-main
- * module owns the upstream source (Ollama library scrape, LM Studio community
- * catalog); the renderer filters/sorts the returned rows locally.
+ * Fetch an engine's populated catalog or submit an explicit upstream query.
+ * Electron main owns every upstream source; the renderer only maps display rows.
*/
-export async function searchEngineHub(engine: EngineType): Promise {
+export async function searchEngineHub(engine: EngineType, query?: string): Promise {
if (!getEngineHub(engine)) return []
- const { models } = await window.pairApi.engines.searchHub(engine)
+ const { models } = await window.pairApi.engines.searchHub(engine, query)
return models.map(hubModelToEntry)
}
diff --git a/desktop/tests/modular/delete-model-restart.test.ts b/desktop/tests/modular/delete-model-restart.test.ts
index 3563c5ee..fbdcef4f 100644
--- a/desktop/tests/modular/delete-model-restart.test.ts
+++ b/desktop/tests/modular/delete-model-restart.test.ts
@@ -24,10 +24,11 @@ import type { EngineType } from '@/shared/types/engines'
const MANIFEST_DIR = path.resolve(process.cwd(), '../services/nvpair-engine-manager/manifests')
-/** Manifest engine ids differ from our `EngineType` for LM Studio only. */
+/** Map engine-manager manifest ids to the desktop's closed engine ids. */
const ENGINE_TYPE_BY_MANIFEST_ID: Record = {
ollama: 'ollama',
- lmstudio: 'lm-studio'
+ lmstudio: 'lm-studio',
+ llamacpp: 'llama-cpp'
}
interface ManifestAction {
@@ -65,6 +66,14 @@ describe('delete-model restart is scoped to LM Studio', () => {
expect(EngineCapabilities.ollama.hasDeleteModel).toBe(true)
})
+ it('llama.cpp deletes without restarting the router', () => {
+ const llamacpp = readManifests().find(m => m.engine === 'llamacpp')
+ expect(llamacpp?.actions?.delete_model).toBeDefined()
+ expect(llamacpp?.actions?.delete_model?.restart_after).toBeUndefined()
+ expect(EngineCapabilities['llama-cpp'].restartsOnModelDelete).toBeFalsy()
+ expect(EngineCapabilities['llama-cpp'].hasDeleteModel).toBe(true)
+ })
+
it('every engine that bounces on delete also warns the user first', () => {
for (const manifest of readManifests()) {
const engineType = ENGINE_TYPE_BY_MANIFEST_ID[manifest.engine]
diff --git a/desktop/tests/modular/engine-command-load.test.ts b/desktop/tests/modular/engine-command-load.test.ts
index 1689cc1d..bf379fef 100644
--- a/desktop/tests/modular/engine-command-load.test.ts
+++ b/desktop/tests/modular/engine-command-load.test.ts
@@ -91,4 +91,42 @@ describe('local model load command', () => {
expect.any(Function)
)
})
+
+ it('routes llama.cpp load and unload through its manifest actions', async () => {
+ await handleServiceBridgeInvoke('engine:command', {
+ command: 'loadModel',
+ engineType: 'llama-cpp',
+ nodeId: 'local-node',
+ model: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M'
+ })
+ await handleServiceBridgeInvoke('engine:command', {
+ command: 'unloadModel',
+ engineType: 'llama-cpp',
+ nodeId: 'local-node',
+ model: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M'
+ })
+
+ expect(mocks.supervisor.sendProcess).toHaveBeenNthCalledWith(
+ 1,
+ 'broker',
+ 'engine:action',
+ {
+ engine: 'llamacpp',
+ action: 'load_model',
+ params: { model: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M' }
+ },
+ expect.any(Function)
+ )
+ expect(mocks.supervisor.sendProcess).toHaveBeenNthCalledWith(
+ 2,
+ 'broker',
+ 'engine:action',
+ {
+ engine: 'llamacpp',
+ action: 'unload_model',
+ params: { model: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M' }
+ },
+ expect.any(Function)
+ )
+ })
})
diff --git a/desktop/tests/modular/engine-identity.test.ts b/desktop/tests/modular/engine-identity.test.ts
new file mode 100644
index 00000000..9456e583
--- /dev/null
+++ b/desktop/tests/modular/engine-identity.test.ts
@@ -0,0 +1,66 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import { describe, expect, it } from 'vitest'
+import {
+ EnabledEngineTypes,
+ EngineDefaultLinks,
+ EngineDisplayNames,
+ EngineManagerNames,
+ EngineTypes
+} from '@/shared/constants/engines'
+import { engineManagerName, engineTypeFromManagerName, isEngineType } from '@/shared/utils/engines'
+import { EngineCapabilities } from '@/ui/constants/engine-capabilities'
+import { getWelcomeEngineCandidates, WELCOME_ENGINE_DEFAULT_SELECTED } from '@/ui/constants/welcome'
+import { gatewayEndpointDisplayUrl } from '@/ui/utils/gateway-inference-paths'
+
+describe('engine identity', () => {
+ it('enables llama.cpp workflows on every platform', () => {
+ expect(EngineTypes).toEqual(['ollama', 'lm-studio', 'llama-cpp'])
+ expect(EnabledEngineTypes).toEqual(['ollama', 'lm-studio', 'llama-cpp'])
+ expect(getWelcomeEngineCandidates('Windows')).toContain('llama-cpp')
+ expect(getWelcomeEngineCandidates('MacOS')).toContain('llama-cpp')
+ expect(getWelcomeEngineCandidates('Linux')).toContain('llama-cpp')
+ })
+
+ it('preselects every engine during onboarding', () => {
+ expect(WELCOME_ENGINE_DEFAULT_SELECTED).toEqual({
+ ollama: true,
+ 'lm-studio': true,
+ 'llama-cpp': true
+ })
+ })
+
+ it('maps the desktop id to the sole engine-manager wire id', () => {
+ expect(EngineManagerNames['llama-cpp']).toBe('llamacpp')
+ expect(engineManagerName('llama-cpp')).toBe('llamacpp')
+ expect(engineTypeFromManagerName('llamacpp')).toBe('llama-cpp')
+ expect(isEngineType('llama-cpp')).toBe(true)
+ })
+
+ it('provides complete display metadata and capabilities', () => {
+ for (const engineType of EngineTypes) {
+ expect(EngineDisplayNames[engineType]).toBeTruthy()
+ expect(EngineDefaultLinks[engineType].docsUrl).toMatch(/^https:\/\//)
+ expect(EngineDefaultLinks[engineType].installUrl).toMatch(/^https:\/\//)
+ expect(EngineCapabilities[engineType]).toBeDefined()
+ }
+
+ expect(EngineDisplayNames['llama-cpp']).toBe('llama.cpp')
+ expect(EngineCapabilities['llama-cpp']).toMatchObject({
+ hasExpiry: false,
+ hasEject: true,
+ hasInstall: ['win32', 'darwin', 'linux'],
+ hasEnginePort: true,
+ hasInstallPath: false,
+ hasProxyWebUI: false,
+ hasPreferredNode: false,
+ hasCrashAlert: false,
+ hasModelSearchOnlyWhenRunning: true,
+ modelOpsWhenStopped: false,
+ hasDeleteModel: true,
+ engineHub: { label: 'llama.cpp' }
+ })
+ expect(gatewayEndpointDisplayUrl(8080, 'llama-cpp')).toBe('http://127.0.0.1:8080')
+ })
+})
diff --git a/desktop/tests/modular/inference-demo-lifecycle.test.ts b/desktop/tests/modular/inference-demo-lifecycle.test.ts
index d9c4b95e..54cb20fb 100644
--- a/desktop/tests/modular/inference-demo-lifecycle.test.ts
+++ b/desktop/tests/modular/inference-demo-lifecycle.test.ts
@@ -2,6 +2,7 @@
// SPDX-License-Identifier: Apache-2.0
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
+import type { EngineType } from '@/shared/types/engines'
/**
* Lifecycle guarantees for the Inference Demo scheduler.
@@ -30,6 +31,7 @@ const spawned: SpawnRecord[] = []
/** Options each `--list-models` discovery probe was invoked with. */
const probeOptions: ChildOptions[] = []
+const probeArgs: string[][] = []
/**
* Poisoned parent environment. `INFERENCE_DISPATCHER_LOOP` would run the child
@@ -45,9 +47,10 @@ const POISONED_ENV: Record = {
}
/** Proxy ports the fake broker reports. Mutable so a test can withhold one. */
-const proxyPorts: Record<'ollama' | 'lm-studio', number | null> = {
+const proxyPorts: Record = {
ollama: 11434,
- 'lm-studio': 1234
+ 'lm-studio': 1234,
+ 'llama-cpp': 8080
}
/** Model inventory each probe returns. Mutable so a test can return none. */
@@ -96,6 +99,7 @@ vi.mock('node:child_process', () => ({
options: ChildOptions,
callback: (error: Error | null, stdout: string, stderr: string) => void
) => {
+ probeArgs.push(_args)
probeOptions.push(options)
callback(null, JSON.stringify(inventory), '')
return { exitCode: null, once: () => {}, kill: () => true }
@@ -114,7 +118,7 @@ vi.mock('electron', () => ({
vi.mock('@/electron/service-bridge/modular-state', () => ({
getModularBridgeState: () => ({
- getProxyPort: (engine: 'ollama' | 'lm-studio') => proxyPorts[engine]
+ getProxyPort: (engine: EngineType) => proxyPorts[engine]
})
}))
@@ -126,18 +130,24 @@ import {
} from '@/electron/inference-demo'
/** Ports the demo is allowed to target: proxy facades only. */
-const PROXY_FACADE_PORTS = [11434, 1234]
+const PROXY_FACADE_PORTS = [11434, 1234, 8080]
function portOf(record: SpawnRecord): number {
return Number(record.args[record.args.indexOf('--port') + 1])
}
+function backendOf(record: SpawnRecord): string {
+ return record.args[record.args.indexOf('--backend') + 1] ?? ''
+}
+
beforeEach(() => {
spawned.length = 0
probeOptions.length = 0
+ probeArgs.length = 0
inventory = [{ name: 'demo-model', type: 'llm' }]
proxyPorts.ollama = 11434
proxyPorts['lm-studio'] = 1234
+ proxyPorts['llama-cpp'] = 8080
Object.assign(process.env, POISONED_ENV)
vi.useFakeTimers()
})
@@ -203,14 +213,30 @@ describe('inference demo lifecycle', () => {
}
})
+ it('probes and schedules llama.cpp through its proxy facade', async () => {
+ await startInferenceDemo()
+ await vi.advanceTimersByTimeAsync(12_000)
+
+ expect(
+ probeArgs.some(
+ args =>
+ args[args.indexOf('--backend') + 1] === 'llamacpp' &&
+ args.includes('--list-models')
+ )
+ ).toBe(true)
+ const llamaCPPRequest = spawned.find(child => backendOf(child) === 'llamacpp')
+ expect(llamaCPPRequest).toBeDefined()
+ if (llamaCPPRequest) expect(portOf(llamaCPPRequest)).toBe(8080)
+ })
+
it('skips an engine whose proxy has not reported a port', async () => {
- proxyPorts['lm-studio'] = null
+ proxyPorts['llama-cpp'] = null
await startInferenceDemo()
await vi.advanceTimersByTimeAsync(70_000)
expect(spawned.length).toBeGreaterThan(0)
for (const child of spawned) {
- expect(portOf(child)).toBe(11434)
+ expect(portOf(child)).not.toBe(8080)
}
})
diff --git a/desktop/tests/modular/inference-demo-schedule.test.ts b/desktop/tests/modular/inference-demo-schedule.test.ts
index 359dcdba..6d93a6be 100644
--- a/desktop/tests/modular/inference-demo-schedule.test.ts
+++ b/desktop/tests/modular/inference-demo-schedule.test.ts
@@ -13,11 +13,17 @@ import {
// Ports are the proxy facades PAIR routes through, never an engine's own
// backend port. At runtime these come from the broker's proxy registry.
function targets(count: number): DemoTarget[] {
- return Array.from({ length: count }, (_, index) => ({
- backend: index % 2 === 0 ? ('ollama' as const) : ('lmstudio' as const),
- port: index % 2 === 0 ? 11434 : 1234,
- model: `model-${index}`
- }))
+ return Array.from({ length: count }, (_, index) => {
+ const model = `model-${index}`
+ switch (index % 3) {
+ case 0:
+ return { backend: 'ollama', port: 11434, model }
+ case 1:
+ return { backend: 'lmstudio', port: 1234, model }
+ default:
+ return { backend: 'llamacpp', port: 8080, model }
+ }
+ })
}
function key(target: DemoTarget): string {
diff --git a/desktop/tests/modular/llamacpp-catalog-cache.test.ts b/desktop/tests/modular/llamacpp-catalog-cache.test.ts
new file mode 100644
index 00000000..4aaca003
--- /dev/null
+++ b/desktop/tests/modular/llamacpp-catalog-cache.test.ts
@@ -0,0 +1,199 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import { afterEach, describe, expect, it, vi } from 'vitest'
+import {
+ LlamaCppCatalogCache,
+ llamaCppPublisherRequestParams,
+ llamaCppSearchRequestParams
+} from '@/electron/model-hub/llamacpp-catalog'
+import type { JsonValue } from '@/shared/types/json'
+
+function hubModel(id: string): JsonValue {
+ return {
+ id,
+ gated: false,
+ private: false,
+ lastModified: '2026-09-20T04:11:57.000Z',
+ downloads: 120,
+ likes: 8,
+ tags: ['gguf', 'text-generation'],
+ pipeline_tag: 'text-generation',
+ siblings: [{ rfilename: 'model-Q4_K_M.gguf' }]
+ }
+}
+
+function deferredJson() {
+ let resolvePromise: (value: JsonValue) => void = () => {
+ throw new Error('deferred JSON promise was not initialized')
+ }
+ const promise = new Promise(resolve => {
+ resolvePromise = resolve
+ })
+ return {
+ promise,
+ resolve: (value: JsonValue) => resolvePromise(value)
+ }
+}
+
+afterEach(() => {
+ vi.useRealTimers()
+})
+
+describe('llama.cpp populated catalog cache', () => {
+ it('builds a bounded Hugging Face request for one publisher', () => {
+ expect(llamaCppPublisherRequestParams('bartowski')).toEqual({
+ author: 'bartowski',
+ filter: 'gguf',
+ sort: 'downloads',
+ direction: -1,
+ limit: 50,
+ full: true
+ })
+ })
+
+ it('builds a bounded all-publisher Hugging Face search request', () => {
+ expect(llamaCppSearchRequestParams('qwen coder')).toEqual({
+ search: 'qwen coder',
+ filter: 'gguf',
+ sort: 'downloads',
+ direction: -1,
+ limit: 50,
+ full: true
+ })
+ })
+
+ it('loads and combines every approved publisher', async () => {
+ const fetchPublisher = vi.fn<(publisher: string) => Promise>()
+ fetchPublisher.mockImplementation(async publisher => [
+ hubModel(`${publisher}/Example-GGUF`)
+ ])
+ const cache = new LlamaCppCatalogCache(fetchPublisher)
+
+ await cache.ensureLoaded()
+
+ expect(fetchPublisher.mock.calls.map(([publisher]) => publisher)).toEqual([
+ 'ggml-org',
+ 'bartowski',
+ 'unsloth'
+ ])
+ expect(cache.list().map(model => model.id)).toEqual([
+ 'ggml-org/Example-GGUF:Q4_K_M',
+ 'bartowski/Example-GGUF:Q4_K_M',
+ 'unsloth/Example-GGUF:Q4_K_M'
+ ])
+ expect(cache.isFetching).toBe(false)
+ expect(cache.size).toBe(3)
+ })
+
+ it('shares one in-flight refresh between callers', async () => {
+ const deferred = deferredJson()
+ const fetchPublisher = vi.fn<(publisher: string) => Promise>()
+ fetchPublisher.mockReturnValue(deferred.promise)
+ const cache = new LlamaCppCatalogCache(fetchPublisher)
+
+ const first = cache.ensureLoaded()
+ const second = cache.ensureLoaded()
+
+ expect(cache.isFetching).toBe(true)
+ expect(fetchPublisher).toHaveBeenCalledTimes(3)
+ deferred.resolve([hubModel('owner/Example-GGUF')])
+ await Promise.all([first, second])
+
+ expect(fetchPublisher).toHaveBeenCalledTimes(3)
+ expect(cache.size).toBe(1)
+ })
+
+ it('keeps the last good catalog when a stale refresh fails', async () => {
+ vi.useFakeTimers()
+ vi.setSystemTime(new Date('2026-09-20T00:00:00.000Z'))
+ const fetchPublisher = vi.fn<(publisher: string) => Promise>()
+ fetchPublisher.mockImplementation(async publisher => [
+ hubModel(`${publisher}/Example-GGUF`)
+ ])
+ const cache = new LlamaCppCatalogCache(fetchPublisher)
+ await cache.ensureLoaded()
+
+ fetchPublisher.mockClear()
+ fetchPublisher.mockRejectedValue(new Error('offline'))
+ vi.setSystemTime(new Date('2026-09-20T07:00:00.000Z'))
+
+ await expect(cache.ensureLoaded()).resolves.toBeUndefined()
+ expect(fetchPublisher).toHaveBeenCalledTimes(3)
+ expect(cache.size).toBe(3)
+ })
+
+ it('surfaces a cold fetch failure', async () => {
+ const fetchPublisher = vi.fn<(publisher: string) => Promise>()
+ fetchPublisher.mockRejectedValue(new Error('offline'))
+ const cache = new LlamaCppCatalogCache(fetchPublisher)
+
+ await expect(cache.ensureLoaded()).rejects.toThrow(
+ 'Unable to load llama.cpp model catalog: offline'
+ )
+ expect(cache.list()).toEqual([])
+ })
+
+ it('searches every public publisher and caches normalized queries', async () => {
+ const fetchPublisher = vi.fn<(publisher: string) => Promise>()
+ const fetchSearch = vi.fn<(query: string) => Promise>()
+ fetchSearch.mockResolvedValue([
+ hubModel('community-author/Qwen-Coder-GGUF'),
+ hubModel('unsloth/Qwen-Coder-GGUF')
+ ])
+ const cache = new LlamaCppCatalogCache(fetchPublisher, fetchSearch)
+
+ const first = await cache.search(' Qwen Coder ')
+ const second = await cache.search('qwen coder')
+
+ expect(fetchSearch).toHaveBeenCalledOnce()
+ expect(fetchSearch).toHaveBeenCalledWith('Qwen Coder')
+ expect(first.map(model => model.id)).toEqual([
+ 'community-author/Qwen-Coder-GGUF:Q4_K_M',
+ 'unsloth/Qwen-Coder-GGUF:Q4_K_M'
+ ])
+ expect(second).toBe(first)
+ expect(fetchPublisher).not.toHaveBeenCalled()
+ })
+
+ it('bounds the process-local search cache', async () => {
+ const fetchPublisher = vi.fn<(publisher: string) => Promise>()
+ const fetchSearch = vi.fn<(query: string) => Promise>()
+ fetchSearch.mockResolvedValue([hubModel('owner/Result-GGUF')])
+ const cache = new LlamaCppCatalogCache(fetchPublisher, fetchSearch)
+
+ for (let index = 0; index < 21; index += 1) {
+ await cache.search(`model-${index}`)
+ }
+ await cache.search('model-0')
+
+ expect(fetchSearch).toHaveBeenCalledTimes(22)
+ })
+
+ it('uses a stale search result when its refresh fails', async () => {
+ vi.useFakeTimers()
+ vi.setSystemTime(new Date('2026-09-20T00:00:00.000Z'))
+ const fetchPublisher = vi.fn<(publisher: string) => Promise>()
+ const fetchSearch = vi.fn<(query: string) => Promise>()
+ fetchSearch.mockResolvedValue([hubModel('owner/Result-GGUF')])
+ const cache = new LlamaCppCatalogCache(fetchPublisher, fetchSearch)
+ const initial = await cache.search('result')
+
+ fetchSearch.mockRejectedValue(new Error('offline'))
+ vi.setSystemTime(new Date('2026-09-20T07:00:00.000Z'))
+
+ await expect(cache.search('result')).resolves.toBe(initial)
+ expect(fetchSearch).toHaveBeenCalledTimes(2)
+ })
+
+ it('surfaces a cold search failure without logging the query', async () => {
+ const fetchPublisher = vi.fn<(publisher: string) => Promise>()
+ const fetchSearch = vi.fn<(query: string) => Promise>()
+ fetchSearch.mockRejectedValue(new Error('offline'))
+ const cache = new LlamaCppCatalogCache(fetchPublisher, fetchSearch)
+
+ await expect(cache.search('private model name')).rejects.toEqual(
+ new Error('Unable to search llama.cpp models: offline')
+ )
+ })
+})
diff --git a/desktop/tests/modular/llamacpp-catalog.test.ts b/desktop/tests/modular/llamacpp-catalog.test.ts
new file mode 100644
index 00000000..db93fbc9
--- /dev/null
+++ b/desktop/tests/modular/llamacpp-catalog.test.ts
@@ -0,0 +1,97 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import { afterEach, describe, expect, it, vi } from 'vitest'
+import { getEngineHubModels } from '@/electron/model-hub'
+import { llamaCppCatalogCache } from '@/electron/model-hub/llamacpp-catalog'
+import type { EngineHubModel } from '@/shared/types/engine-api'
+import { isHubEntryDownloaded } from '@/ui/utils/match-downloaded-model'
+import type { ModelItem } from '@/ui/types/engine-info'
+import type { ModelEntry } from '@/ui/types/model-hub'
+
+function modelEntry(id: string): ModelEntry {
+ return {
+ id,
+ name: id,
+ author: id.slice(0, id.indexOf('/')),
+ url: `https://huggingface.co/${id.slice(0, id.lastIndexOf(':'))}`,
+ updatedAt: new Date(0)
+ }
+}
+
+function modelItem(name: string, downloaded = true): ModelItem {
+ return {
+ name,
+ size: 0,
+ downloaded,
+ status: 'idle',
+ parameterSize: '',
+ quantization: '',
+ family: '',
+ digest: '',
+ sizeVram: null,
+ expiresAt: null,
+ expiry: '10m',
+ capabilities: []
+ }
+}
+
+afterEach(() => {
+ vi.restoreAllMocks()
+})
+
+describe('live llama.cpp catalog', () => {
+ it('serves the warmed catalog without transforming pull keys', async () => {
+ const models: EngineHubModel[] = [
+ {
+ id: 'owner/model:Q4_K_M',
+ name: 'owner/model:Q4_K_M',
+ author: 'owner',
+ url: 'https://huggingface.co/owner/model',
+ downloads: 10,
+ likes: 2,
+ updatedAt: '2026-09-20T04:11:57.000Z',
+ tags: ['gguf', 'Q4_K_M']
+ }
+ ]
+ const ensureLoaded = vi
+ .spyOn(llamaCppCatalogCache, 'ensureLoaded')
+ .mockResolvedValue(undefined)
+ vi.spyOn(llamaCppCatalogCache, 'list').mockReturnValue(models)
+
+ await expect(getEngineHubModels('llama-cpp')).resolves.toEqual({ models })
+ expect(ensureLoaded).toHaveBeenCalledOnce()
+ })
+
+ it('routes an explicit query through the bounded search cache', async () => {
+ const models: EngineHubModel[] = [
+ {
+ id: 'community/model:Q4_K_M',
+ name: 'community/model:Q4_K_M',
+ author: 'community',
+ url: 'https://huggingface.co/community/model',
+ downloads: 5,
+ likes: 1,
+ updatedAt: '2026-09-20T04:11:57.000Z',
+ tags: ['gguf', 'Q4_K_M']
+ }
+ ]
+ const search = vi.spyOn(llamaCppCatalogCache, 'search').mockResolvedValue(models)
+ const ensureLoaded = vi.spyOn(llamaCppCatalogCache, 'ensureLoaded')
+
+ await expect(getEngineHubModels('llama-cpp', 'coder')).resolves.toEqual({ models })
+ expect(search).toHaveBeenCalledWith('coder')
+ expect(ensureLoaded).not.toHaveBeenCalled()
+ })
+
+ it('hides only an exact downloaded router model id', () => {
+ const id = 'owner/model:Q4_K_M'
+ const entry = modelEntry(id)
+
+ expect(isHubEntryDownloaded('llama-cpp', entry, [modelItem(id)])).toBe(true)
+ expect(
+ isHubEntryDownloaded('llama-cpp', entry, [modelItem(id.slice(0, id.lastIndexOf(':')))])
+ ).toBe(false)
+ expect(isHubEntryDownloaded('llama-cpp', entry, [modelItem(id, false)])).toBe(false)
+ })
+})
diff --git a/desktop/tests/modular/llamacpp-huggingface.test.ts b/desktop/tests/modular/llamacpp-huggingface.test.ts
new file mode 100644
index 00000000..2a054b9c
--- /dev/null
+++ b/desktop/tests/modular/llamacpp-huggingface.test.ts
@@ -0,0 +1,112 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import { describe, expect, it } from 'vitest'
+import { normalizeLlamaCppHuggingFaceModels } from '@/electron/model-hub/llamacpp-huggingface'
+import type { JsonValue } from '@/shared/types/json'
+
+interface ModelOptions {
+ gated?: boolean | string
+ isPrivate?: boolean
+ tags?: string[]
+ pipelineTag?: string
+ updatedAt?: string
+}
+
+function hubModel(id: string, files: string[], options: ModelOptions = {}): JsonValue {
+ return {
+ id,
+ gated: options.gated ?? false,
+ private: options.isPrivate ?? false,
+ lastModified: options.updatedAt ?? '2026-09-20T04:11:57.000Z',
+ downloads: 120,
+ likes: 8,
+ tags: options.tags ?? ['gguf', 'text-generation'],
+ pipeline_tag: options.pipelineTag ?? 'text-generation',
+ siblings: files.map(rfilename => ({ rfilename }))
+ }
+}
+
+describe('llama.cpp Hugging Face normalization', () => {
+ it('normalizes a public generative repository into an exact pull key', () => {
+ const [model] = normalizeLlamaCppHuggingFaceModels([
+ hubModel('owner/Model-GGUF', ['Model-Q4_K_M.gguf'])
+ ])
+
+ expect(model).toEqual({
+ id: 'owner/Model-GGUF:Q4_K_M',
+ name: 'owner/Model-GGUF:Q4_K_M',
+ author: 'owner',
+ url: 'https://huggingface.co/owner/Model-GGUF',
+ downloads: 120,
+ likes: 8,
+ updatedAt: '2026-09-20T04:11:57.000Z',
+ tags: ['gguf', 'text-generation', 'Q4_K_M']
+ })
+ })
+
+ it('accepts the first shard of a conversational Q4_K_M model', () => {
+ const models = normalizeLlamaCppHuggingFaceModels([
+ hubModel(
+ 'org/Vision-Chat-GGUF',
+ [
+ 'Q4_K_M/Vision-Chat-Q4_K_M-00001-of-00003.gguf',
+ 'Q4_K_M/Vision-Chat-Q4_K_M-00002-of-00003.gguf'
+ ],
+ { tags: ['gguf', 'conversational'], pipelineTag: 'image-text-to-text' }
+ )
+ ])
+
+ expect(models.map(model => model.id)).toEqual(['org/Vision-Chat-GGUF:Q4_K_M'])
+ })
+
+ it('rejects unsafe, private, and gated repositories', () => {
+ const models = normalizeLlamaCppHuggingFaceModels([
+ hubModel('../owner/model', ['model-Q4_K_M.gguf']),
+ hubModel('owner/model/extra', ['model-Q4_K_M.gguf']),
+ hubModel('owner/private-model', ['model-Q4_K_M.gguf'], { isPrivate: true }),
+ hubModel('owner/gated-model', ['model-Q4_K_M.gguf'], { gated: 'manual' })
+ ])
+
+ expect(models).toEqual([])
+ })
+
+ it('rejects helper-only, wrong-quantization, and incomplete split artifacts', () => {
+ const models = normalizeLlamaCppHuggingFaceModels([
+ hubModel('owner/mmproj', ['mmproj-model-Q4_K_M.gguf']),
+ hubModel('owner/imatrix', ['model-imatrix-Q4_K_M.gguf']),
+ hubModel('owner/mtp', ['model-mtp-Q4_K_M.gguf']),
+ hubModel('owner/q8', ['model-Q8_0.gguf']),
+ hubModel('owner/later-shard', ['model-Q4_K_M-00002-of-00003.gguf'])
+ ])
+
+ expect(models).toEqual([])
+ })
+
+ it('rejects non-GGUF, non-generative, and invalid-date metadata', () => {
+ const models = normalizeLlamaCppHuggingFaceModels([
+ hubModel('owner/no-gguf-tag', ['model-Q4_K_M.gguf'], {
+ tags: ['text-generation']
+ }),
+ hubModel('owner/embedding', ['model-Q4_K_M.gguf'], {
+ tags: ['gguf'],
+ pipelineTag: 'feature-extraction'
+ }),
+ hubModel('owner/no-date', ['model-Q4_K_M.gguf'], {
+ updatedAt: 'not-a-date'
+ })
+ ])
+
+ expect(models).toEqual([])
+ })
+
+ it('deduplicates repeated repository metadata', () => {
+ const models = normalizeLlamaCppHuggingFaceModels([
+ hubModel('owner/model', ['model-Q4_K_M.gguf']),
+ hubModel('owner/model', ['model-Q4_K_M.gguf'])
+ ])
+
+ expect(models).toHaveLength(1)
+ expect(models[0]?.id).toBe('owner/model:Q4_K_M')
+ })
+})
diff --git a/desktop/tests/modular/lmstudio-stale-model.test.ts b/desktop/tests/modular/lmstudio-stale-model.test.ts
index 8b37e862..23a79fa9 100644
--- a/desktop/tests/modular/lmstudio-stale-model.test.ts
+++ b/desktop/tests/modular/lmstudio-stale-model.test.ts
@@ -12,9 +12,38 @@ vi.mock('@/electron/window', () => ({ createOverviewWindow: vi.fn() }))
import { getModularBridgeState } from '@/electron/service-bridge/modular-state'
import {
getModularSupervisor,
- parseListModelNames
+ parseListModelNames,
+ pullModelParams
} from '@/electron/service-bridge/modular-supervisor'
+describe('engine model wire formats', () => {
+ it('parses strict llama.cpp router inventories', () => {
+ expect(
+ parseListModelNames({
+ data: [
+ { id: 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M', status: { value: 'loaded' } },
+ { id: 'bartowski/Qwen2.5-3B-Instruct-GGUF:Q4_K_M' }
+ ]
+ })
+ ).toEqual([
+ 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M',
+ 'bartowski/Qwen2.5-3B-Instruct-GGUF:Q4_K_M'
+ ])
+ expect(parseListModelNames({ data: [] })).toEqual([])
+ expect(() => parseListModelNames({ data: null })).toThrow('missing its model array')
+ expect(() => parseListModelNames({ data: [{}] })).toThrow('no usable model names')
+ expect(() =>
+ parseListModelNames({ models: null, data: [{ id: 'must-not-mask-malformed-models' }] })
+ ).toThrow('missing its model array')
+ })
+
+ it('uses the manifest model field for llama.cpp pulls', () => {
+ expect(pullModelParams('ollama', 'demo')).toEqual({ name: 'demo' })
+ expect(pullModelParams('lmstudio', 'demo')).toEqual({ model: 'demo' })
+ expect(pullModelParams('llamacpp', 'demo')).toEqual({ model: 'demo' })
+ })
+})
+
describe('LM Studio model reconciliation', () => {
it('parses the native inventory and distinguishes explicit empty from unknown', () => {
expect(
diff --git a/desktop/tests/modular/model-hub-bridge.test.ts b/desktop/tests/modular/model-hub-bridge.test.ts
new file mode 100644
index 00000000..d6397e3f
--- /dev/null
+++ b/desktop/tests/modular/model-hub-bridge.test.ts
@@ -0,0 +1,46 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import { beforeEach, describe, expect, it, vi } from 'vitest'
+import type { getEngineHubModels } from '@/electron/model-hub'
+
+const mocks = vi.hoisted(() => ({
+ getEngineHubModels: vi.fn()
+}))
+
+vi.mock('@/electron/service-bridge/modular-supervisor', () => ({
+ getModularSupervisor: () => ({})
+}))
+vi.mock('@/electron/service-bridge/modular-state', () => ({
+ getModularBridgeState: () => ({}),
+ isProxyEngine: () => false,
+ isUpstreamUnreachableError: () => false,
+ parseServiceErrors: () => []
+}))
+vi.mock('@/electron/model-hub', () => ({
+ getEngineHubModels: mocks.getEngineHubModels
+}))
+
+import { handleServiceBridgeInvoke } from '@/electron/service-bridge/empty-handlers'
+
+describe('model hub service bridge', () => {
+ beforeEach(() => {
+ vi.clearAllMocks()
+ mocks.getEngineHubModels.mockResolvedValue({ models: [] })
+ })
+
+ it('forwards an explicit llama.cpp query to Electron main', async () => {
+ await handleServiceBridgeInvoke('engine:search-hub', {
+ engineType: 'llama-cpp',
+ query: 'qwen coder'
+ })
+
+ expect(mocks.getEngineHubModels).toHaveBeenCalledWith('llama-cpp', 'qwen coder')
+ })
+
+ it('preserves populated-catalog requests without a query', async () => {
+ await handleServiceBridgeInvoke('engine:search-hub', { engineType: 'llama-cpp' })
+
+ expect(mocks.getEngineHubModels).toHaveBeenCalledWith('llama-cpp', undefined)
+ })
+})
diff --git a/desktop/tests/modular/model-hub-search.test.ts b/desktop/tests/modular/model-hub-search.test.ts
new file mode 100644
index 00000000..a13b906f
--- /dev/null
+++ b/desktop/tests/modular/model-hub-search.test.ts
@@ -0,0 +1,48 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import { describe, expect, it } from 'vitest'
+import { mergeModelHubResults } from '@/ui/utils/model-hub-search'
+import type { ModelEntry } from '@/ui/types/model-hub'
+
+function model(id: string, updatedAt: string): ModelEntry {
+ return {
+ id,
+ name: id,
+ author: id.slice(0, id.indexOf('/')),
+ url: `https://huggingface.co/${id.slice(0, id.lastIndexOf(':'))}`,
+ updatedAt: new Date(updatedAt)
+ }
+}
+
+describe('model hub result merging', () => {
+ it('appends remote matches while preserving populated order', () => {
+ const populated = [
+ model('approved/alpha:Q4_K_M', '2026-01-01T00:00:00.000Z'),
+ model('approved/shared:Q4_K_M', '2026-01-01T00:00:00.000Z')
+ ]
+ const remote = [
+ model('approved/shared:Q4_K_M', '2026-09-01T00:00:00.000Z'),
+ model('community/beta:Q4_K_M', '2026-08-01T00:00:00.000Z')
+ ]
+
+ const merged = mergeModelHubResults(populated, remote)
+
+ expect(merged.map(entry => entry.id)).toEqual([
+ 'approved/alpha:Q4_K_M',
+ 'approved/shared:Q4_K_M',
+ 'community/beta:Q4_K_M'
+ ])
+ expect(merged[1]).toBe(remote[0])
+ })
+
+ it('does not mutate either source list', () => {
+ const populated = [model('approved/alpha:Q4_K_M', '2026-01-01T00:00:00.000Z')]
+ const remote = [model('community/beta:Q4_K_M', '2026-08-01T00:00:00.000Z')]
+
+ mergeModelHubResults(populated, remote)
+
+ expect(populated.map(entry => entry.id)).toEqual(['approved/alpha:Q4_K_M'])
+ expect(remote.map(entry => entry.id)).toEqual(['community/beta:Q4_K_M'])
+ })
+})
diff --git a/desktop/tests/modular/proxy-engine-identity.test.ts b/desktop/tests/modular/proxy-engine-identity.test.ts
new file mode 100644
index 00000000..69031984
--- /dev/null
+++ b/desktop/tests/modular/proxy-engine-identity.test.ts
@@ -0,0 +1,195 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import { describe, expect, it, vi } from 'vitest'
+
+vi.mock('electron', () => ({ BrowserWindow: { getAllWindows: () => [] } }))
+vi.mock('@/electron/window', () => ({ createOverviewWindow: vi.fn() }))
+
+import { getModularBridgeState } from '@/electron/service-bridge/modular-state'
+import {
+ isProxyEngine,
+ PROXY_ENGINES,
+ PROXY_NODE_SOURCES,
+ proxyEngineFromManagerId,
+ proxyEngineFromSource,
+ proxySourceForEngine
+} from '@/electron/service-bridge/proxy-engines'
+import { engineManagerName } from '@/shared/utils/engines'
+
+describe('proxy engine identity', () => {
+ it('declares the complete proxy set in stable order', () => {
+ expect(PROXY_ENGINES).toEqual(['ollama', 'lm-studio', 'llama-cpp'])
+ expect(PROXY_NODE_SOURCES).toEqual(['ollama-proxy', 'lmstudio-proxy', 'llamacpp-proxy'])
+ expect(isProxyEngine('ollama')).toBe(true)
+ expect(isProxyEngine('lm-studio')).toBe(true)
+ expect(isProxyEngine('llama-cpp')).toBe(true)
+ })
+
+ it('round-trips engine, manager, and relay source identities', () => {
+ for (const engine of PROXY_ENGINES) {
+ const source = proxySourceForEngine(engine)
+ expect(proxyEngineFromSource(source)).toBe(engine)
+ expect(proxyEngineFromManagerId(engineManagerName(engine))).toBe(engine)
+ }
+ expect(proxyEngineFromSource('other-proxy')).toBeNull()
+ expect(proxyEngineFromManagerId('other')).toBeNull()
+ expect(proxyEngineFromManagerId('llamacpp')).toBe('llama-cpp')
+ })
+
+ it('routes each proxy notification to the mapped engine', () => {
+ const state = getModularBridgeState()
+ const nodeId = 'proxy-identity-remote'
+ state.setSelfId('proxy-identity-self')
+
+ state.handleNotification({
+ source: 'ollama-proxy',
+ method: 'node/discovered',
+ params: {
+ id: nodeId,
+ host: 'proxy-identity-host',
+ port: 11434,
+ addresses: ['192.0.2.40'],
+ ip: '192.0.2.40'
+ }
+ })
+ state.handleNotification({
+ source: 'lmstudio-proxy',
+ method: 'node/discovered',
+ params: {
+ id: nodeId,
+ host: 'proxy-identity-host',
+ port: 1234,
+ addresses: ['192.0.2.40'],
+ ip: '192.0.2.40'
+ }
+ })
+ state.handleNotification({
+ source: 'llamacpp-proxy',
+ method: 'ready',
+ params: { port: 8080 }
+ })
+ state.handleNotification({
+ source: 'llamacpp-proxy',
+ method: 'node/discovered',
+ params: {
+ id: nodeId,
+ host: 'proxy-identity-host',
+ port: 8080,
+ addresses: ['192.0.2.40'],
+ ip: '192.0.2.40'
+ }
+ })
+ state.handleNotification({
+ source: 'broker',
+ method: 'discovery:nodes-changed',
+ params: {
+ nodes: [
+ {
+ hostUuid: nodeId,
+ name: 'proxy-identity-host',
+ ipAddress: '192.0.2.40',
+ port: 14318,
+ models: ['owner/alpha:Q4_K_M', 'owner/beta:Q4_K_M'],
+ modelsByEngine: {
+ llamacpp: ['owner/alpha:Q4_K_M', 'owner/beta:Q4_K_M']
+ },
+ loadedByEngine: { llamacpp: ['owner/alpha:Q4_K_M'] }
+ }
+ ]
+ }
+ })
+ state.applyRemoteEngineFacts(nodeId, {
+ engines: [
+ {
+ engine: 'llamacpp',
+ installed: true,
+ running: true,
+ healthy: true,
+ port: 8081
+ }
+ ]
+ })
+
+ const statuses = state
+ .getEngineInitialState()
+ .statuses.filter(status => status.nodeId === nodeId)
+ expect(statuses).toEqual([
+ expect.objectContaining({
+ engineType: 'ollama',
+ processStatus: 'running',
+ proxyPort: 11434
+ }),
+ expect.objectContaining({
+ engineType: 'lm-studio',
+ processStatus: 'running',
+ proxyPort: 1234
+ }),
+ expect.objectContaining({
+ engineType: 'llama-cpp',
+ processStatus: 'running',
+ enginePort: 8081,
+ proxyPort: 8080
+ })
+ ])
+
+ const llamaModels = state
+ .getEngineInitialState()
+ .models.find(models => models.nodeId === nodeId && models.engineType === 'llama-cpp')
+ expect(llamaModels?.models).toEqual([
+ expect.objectContaining({ name: 'owner/alpha:Q4_K_M', status: 'loaded' }),
+ expect.objectContaining({ name: 'owner/beta:Q4_K_M', status: 'idle' })
+ ])
+ })
+
+ it('projects local llama.cpp lifecycle and residency into the initial snapshot', () => {
+ const state = getModularBridgeState()
+ const nodeId = 'proxy-identity-local'
+ const model = 'ggml-org/gemma-3-1b-it-GGUF:Q4_K_M'
+ state.setSelfId(nodeId)
+ state.handleNotification({
+ source: 'broker',
+ method: 'discovery:nodes-changed',
+ params: {
+ nodes: [
+ {
+ hostUuid: nodeId,
+ name: 'proxy-identity-local-host',
+ ipAddress: '127.0.0.1',
+ port: 14318
+ }
+ ]
+ }
+ })
+ state.handleNotification({
+ source: 'llamacpp-proxy',
+ method: 'ready',
+ params: { port: 8080 }
+ })
+ state.applyEngineManagerStatus({
+ engine: 'llamacpp',
+ installed: true,
+ running: true,
+ healthy: true,
+ port: 8081
+ })
+ state.setLocalEngineModels('llama-cpp', [model])
+ state.applyLocalLoadedModels({ llamacpp: [model] })
+
+ const initial = state.getEngineInitialState()
+ expect(
+ initial.statuses.find(
+ status => status.nodeId === nodeId && status.engineType === 'llama-cpp'
+ )
+ ).toMatchObject({
+ processStatus: 'running',
+ enginePort: 8081,
+ proxyPort: 8080
+ })
+ expect(
+ initial.models.find(
+ models => models.nodeId === nodeId && models.engineType === 'llama-cpp'
+ )?.models
+ ).toEqual([expect.objectContaining({ name: model, status: 'loaded' })])
+ })
+})
diff --git a/desktop/tests/modular/service-startup-failure.test.ts b/desktop/tests/modular/service-startup-failure.test.ts
index 272b45e6..56e8850a 100644
--- a/desktop/tests/modular/service-startup-failure.test.ts
+++ b/desktop/tests/modular/service-startup-failure.test.ts
@@ -91,17 +91,18 @@ describe('service startup failure handling', () => {
it('surfaces a readiness timeout and recovers if readiness arrives later', async () => {
const timeout = new mocks.StartupTimeoutError(
- 'NVIDIA PAIR service did not become ready within 15 seconds'
+ 'NVIDIA PAIR service did not become ready within 90 seconds'
)
mocks.supervisor.waitUntilReady.mockRejectedValueOnce(timeout)
await expect(initializeConnector()).rejects.toThrow('service did not become ready')
+ expect(mocks.supervisor.waitUntilReady).toHaveBeenCalledWith(90_000)
expect(getConnectorStatus()).toBe('reconnecting')
expect(didWeSpawnCli()).toBe(true)
expect(getConnectorError()).toContain('service did not become ready')
expect(mocks.notifyBrokerStartupFailure).toHaveBeenCalledWith(
- 'NVIDIA PAIR service did not become ready within 15 seconds'
+ 'NVIDIA PAIR service did not become ready within 90 seconds'
)
mocks.invokeReady()
diff --git a/docs/architecture.mdx b/docs/architecture.mdx
index b1c50e79..dcb05cb3 100644
--- a/docs/architecture.mdx
+++ b/docs/architecture.mdx
@@ -40,9 +40,9 @@ flowchart TB
one cluster.
- A **node** is one machine. Nodes are peers where each runs the same services, and
each can both serve requests and route them elsewhere.
-- An **engine** is an inference server on a node, Ollama or LM Studio. A node can
- run both, one, or neither, and a node with no running engine is not eligible to
- serve.
+- An **engine** is an inference server on a node: Ollama, LM Studio, or
+ llama.cpp. A node can run any combination, and a node with no running engine
+ is not eligible to serve.
- **Models** belong to an engine on a specific node. Nothing is shared. The same
model on two nodes is two independent copies and that duplication is what makes
the two nodes interchangeable for a request.
@@ -314,8 +314,8 @@ ordering reaches a proxy in three steps:
The ranking combines *pending work and GPU pressure*. A workload counts as
pending while it is queued or running, and it is attributed to the node it was
-placed on. Both engines count together, so Ollama load affects LM Studio ordering
-and vice versa.
+placed on. Every engine counts in the same node-wide pool, so work through one
+facade affects the others' ordering.
GPU pressure is deliberately coarse. The scheduler smooths the busiest GPU's
utilization, maps it to 0–3 pressure units at 40%, 70%, and 85%, and uses lower
@@ -366,9 +366,10 @@ advertised owner eligible.
A request whose model cannot be parsed keeps the ordinary non-model ordering.
-Model listings are not routed at all. A `GET` of `/v1/models` or `/api/tags` is
-fanned out to every candidate concurrently and the replies are merged, which is
-why the answer is the cluster's inventory rather than one node's.
+Model listings are not routed at all. A `GET` of `/v1/models`, Ollama's
+`/api/tags`, or llama.cpp's `/models` is fanned out to every candidate for that
+engine concurrently and the replies are merged, which is why the answer is the
+cluster's inventory for that engine rather than one node's.
#### Failover and Inventory Freshness
@@ -428,8 +429,8 @@ one place lower in the order.
to an engine's own port is absent from workload events. GPU-heavy external work
can still raise pressure, but CPU-only work and queued demand remain invisible.
-**Both engines are counted as one pool.** Ollama and LM Studio load is summed,
-and maximum GPU pressure applies to the whole node. That is conservative on a
+**All engines are counted as one pool.** Ollama, LM Studio, and llama.cpp load is
+summed, and maximum GPU pressure applies node-wide. This is conservative on a
typical single-GPU machine and can underuse a multi-GPU node where the engines
occupy different devices.
@@ -466,6 +467,11 @@ known install locations for each engine, so an Ollama or LM Studio you installed
yourself is found where it already is. "Installing" an engine that is already
present downloads nothing and reports it as installed.
+llama.cpp uses PAIR's managed router rather than an arbitrary existing server.
+Its readiness check requires the listener to identify itself as a router, so a
+single-model `llama-server` on the configured port is rejected instead of
+adopted. Its model cache is separate and survives uninstall or reinstall.
+
Starting is similarly deferential. If something is already serving the engine's
port, PAIR **adopts** that instance instead of spawning a second copy, and reports
it as running even though it did not start it.
@@ -483,8 +489,8 @@ An application that already works with Ollama is configured for `11434`. If PAIR
listened somewhere else, every tool would need reconfiguring to gain anything, so
PAIR inverts it: the **proxy** takes the port the engine would normally use, and
the engine moves behind it — Ollama to `11435` and upwards, LM Studio to `1235`
-and upwards. Existing clients keep working untouched and transparently gain the
-cluster.
+and upwards. The managed llama.cpp router stays on `8081` while its facade takes
+`8080`. Existing clients keep working untouched and transparently gain the cluster.
This is also what makes the engine unreachable from outside. Engines PAIR starts
bind to loopback, so the only network-facing listener is the proxy, which is where
@@ -492,7 +498,7 @@ cluster authentication lives.
An inherited `OLLAMA_HOST` naming a different local plaintext port is honored as
well. The broker gives the proxy that normalized loopback-only alias, reserved
-against every engine and both proxies' port plans so a relocating engine can never
+against every engine and every facade's port plan so a relocating engine can never
land on it. `localhost` claims IPv4 and IPv6 together, and remote and HTTPS
targets are never intercepted. Clients already configured through the variable
therefore enter the same routing path without being reconfigured.
@@ -522,6 +528,7 @@ A default installation listens on these ports:
| --- | --- |
| `11434` | Ollama-compatible proxy (Ollama itself moves to `11435`+) |
| `1234` | OpenAI-compatible proxy (LM Studio moves to `1235`+) |
+| `8080` | OpenAI-compatible llama.cpp proxy (managed router on `8081`) |
| `14318` | Node hardware and model inventory |
| `14319` | Service-error synchronization between nodes |
| `14320` | Workload propagation between nodes |
@@ -550,7 +557,7 @@ browsing themselves.
The scanner advertises one `_nvpair-node._tcp` multicast DNS (mDNS) record for
each host, and that record carries two kinds of content:
-- The ports its sibling services registered: node-info, both proxies, errors,
+- The ports its sibling services registered: node-info, engine proxies, errors,
workloads, cluster manager, and engine manager
- The node's identity: `uuid=`, `cluster-uuid=` after clustering, and where to
reach it — `ip=` for the address the node ranks first, and `ips=` for the whole
diff --git a/docs/building.mdx b/docs/building.mdx
index 39a232cf..74e6595d 100644
--- a/docs/building.mdx
+++ b/docs/building.mdx
@@ -23,6 +23,16 @@ Install these first:
- [jq](https://jqlang.github.io/jq/download/) on your `PATH`. Install it with
`sudo apt install jq`, `sudo dnf install jq`, `brew install jq`, or
`winget install jqlang.jq`.
+- On Windows, install the
+ [latest Microsoft Visual C++ v14 x64 Redistributable](https://aka.ms/vc14/vc_redist.x64.exe)
+ before using PAIR-managed llama.cpp from a source or services-only run.
+ Release installers install it automatically. Microsoft's x64 package supplies
+ both the x64 and ARM64 runtime binaries.
+
+The reference Windows installer builds stage Microsoft's unmodified
+Redistributable package. Redistributing that package is limited to licensed
+Visual Studio users and remains subject to the
+[Microsoft Software License Terms](https://learn.microsoft.com/en-us/cpp/windows/redistributing-visual-cpp-files).
## Quick Start from Source
diff --git a/docs/engine-lifecycle.mdx b/docs/engine-lifecycle.mdx
index 0d492277..e673abe1 100644
--- a/docs/engine-lifecycle.mdx
+++ b/docs/engine-lifecycle.mdx
@@ -6,10 +6,10 @@ SPDX-License-Identifier: Apache-2.0
# Managing Engines in NVIDIA Personal AI Router
An **engine** is the local inference runtime Personal AI Router (PAIR) uses to
-run models. Today that means Ollama or LM Studio on a given machine. PAIR can
-install and run those engines for you, or work with a copy you already have.
-This page explains what you can expect when you install, start, stop, update, or
-remove an engine.
+run models. Today that means Ollama, LM Studio, or llama.cpp on a given machine.
+PAIR can install and run those engines for you, or work with a supported
+existing Ollama or LM Studio installation. This page explains what you can
+expect when you install, start, stop, update, or remove an engine.
For first-time setup, refer to [Getting started](getting-started.mdx). To change
an engine's ports or launch command, refer to
@@ -44,7 +44,7 @@ operations run.
These actions need the UI:
-- Listing, loading, ejecting, and deleting models
+- Deleting models
- Updating an engine
- Managing another node's engines
@@ -59,6 +59,7 @@ handles these engine lifecycle actions:
- Restart an engine.
- Uninstall an engine.
- Pull a model.
+- List, load, and unload models.
It does not do everything that the application does.
@@ -79,6 +80,17 @@ When you install an engine, consider the following:
- If Ollama or LM Studio is already running on the machine, PAIR can **adopt**
that install instead of downloading another copy. Adoption helps when the
usual engine port is already in use.
+- llama.cpp is always a PAIR-managed install. Windows and Linux download fixed
+ checksum-pinned CUDA server/runtime archives; macOS downloads the standard
+ Metal-capable archive. The roughly 0.6–0.8 GiB engine download happens on
+ demand and is not part of the PAIR installer. Package choice is not
+ driver-aware; llama.cpp retains CPU fallback when GPU acceleration is
+ unavailable.
+
+PAIR-managed llama.cpp models sleep after five minutes without inference work.
+This releases model and KV-cache memory, and the next request wakes the model
+automatically. The llama.cpp worker remains running and may retain some GPU
+memory.
To download models:
@@ -117,7 +129,9 @@ Both actions apply only to an engine PAIR installed:
Uninstalling an engine does **not** remove its models. Downloaded model files stay
on disk, in the engine's own storage such as `~/.ollama`, so uninstalling and
reinstalling does not cost you re-downloading them. To reclaim that space, delete
-the models through the engine, or remove its data directory yourself.
+the models through the engine, or remove its data directory yourself. llama.cpp
+uses a persistent cache beside PAIR's managed install; use **Delete** to remove
+individual entries without removing the engine.
Port changes for an NVPAIR-installed engine also live in **Engine settings**.
Prefer the controls in PAIR over editing the engine's own config when PAIR is
@@ -188,9 +202,9 @@ brings it back without fetching it again.

On a headless machine the terminal interface covers installing, starting,
-stopping, restarting, and uninstalling an engine, and pulling a model. Operations
-it does not have, such as deleting a model or updating an engine, need the
-desktop application on that machine.
+stopping, restarting, and uninstalling an engine; pulling a model; and listing,
+loading, or unloading models. Operations it does not have, such as deleting a
+model or updating an engine, need the desktop application on that machine.
PAIR does **not** warm models. Starting an engine does not pre-load models into
memory, and it holds nothing ready in advance. Depending on the engine, a model
diff --git a/docs/engine-settings.mdx b/docs/engine-settings.mdx
index 61ceb7e1..62a62611 100644
--- a/docs/engine-settings.mdx
+++ b/docs/engine-settings.mdx
@@ -5,9 +5,10 @@ SPDX-License-Identifier: Apache-2.0
# Engine Settings in NVIDIA Personal AI Router
-Open a device's Ollama or LM Studio row and expand **Settings**. Edit the server
-port, proxy port and engine arguments, then select the green **Apply** button. A paired device
-that supports settings can be edited from another device in the cluster.
+Open a device's Ollama, LM Studio, or llama.cpp row and expand **Settings**.
+Edit the server port, proxy port and engine arguments, then select the green
+**Apply** button. A paired device that supports settings can be edited from
+another device in the cluster.
PAIR fixes the executable, lifecycle commands and loopback address. Unsupported
vendor options can fail at startup.
@@ -55,11 +56,13 @@ Other environment assignments can also be edited from a paired device.
Origin lists need a scheme and a specific host, so bare `localhost` and `*` are
rejected, including wildcards hidden inside extra quotes. Wildcard subdomains
such as `https://*.example.com` and ports such as `http://localhost:*` are supported.
-List the origins you want to allow. This is
-how you give a web page access to your models: PAIR's local endpoints add no
-permissive default of their own, so a browser reaches a model through PAIR only
-where the engine would have allowed it directly. For a request that spans the
-cluster, every engine that answers has to allow the origin.
+Managed llama.cpp starts with an empty origin list, so cross-origin browser
+access is disabled until specific origins are added locally. List only the
+origins you want to allow. This is how you give a web page access to your models:
+PAIR's local endpoints add no permissive default of their own, so a browser
+reaches a model through PAIR only where the engine would have allowed it directly.
+For a request that spans the cluster, every engine that answers has to allow the
+origin.
An engine does not need a CORS launch option to work with PAIR. The proxy follows
the engine's HTTP responses regardless of how its CORS policy is configured.
@@ -78,6 +81,7 @@ configuration consistent:
| --- | --- | --- | --- |
| Ollama | `OLLAMA_HOST=127.0.0.1:` | The host in `OLLAMA_HOST` | `OLLAMA_ORIGINS` |
| LM Studio | `--port`, `-p` | `--bind`, `LMS_SERVER_HOST` | `--cors` |
+| llama.cpp | `--port` | `--host` | `--cors-origins` |
For LM Studio, `--port 1235`, `--port=1235`, `-p 1235`, `-p1235` and `-p=1235`
all synchronize the same server-port field. Conflicting repetitions and invalid
@@ -85,6 +89,8 @@ ports are rejected. Write managed short options separately; bundles containing
them are ambiguous and rejected. CORS switches do not take `=true` or `=false`:
add or remove the switch locally. Origin lists are normalized before validation;
conflicting repeated lists are rejected. Other vendor options remain opaque.
+llama.cpp must remain in router mode; model-selection arguments such as `-m`
+or `-hf` switch it to an incompatible single-model server and fail readiness.
Changing the server-port field updates the port in the command. Editing the
command's port updates the numeric field. If both are edited before validation
diff --git a/docs/getting-started.mdx b/docs/getting-started.mdx
index 57c03ffe..d27b362c 100644
--- a/docs/getting-started.mdx
+++ b/docs/getting-started.mdx
@@ -33,8 +33,9 @@ inference; two or more on the same local network let you try pairing and routing
Nothing else has to be in place first:
-- **Engines.** PAIR installs and starts Ollama or LM Studio for you in step 4. If
- an engine is already installed, PAIR detects and uses it instead.
+- **Engines.** PAIR installs and starts Ollama, LM Studio, or llama.cpp for you
+ in step 4. Existing Ollama and LM Studio installations can also be detected
+ and used.
- **Models.** PAIR downloads models for you in step 4. A request needs only one
eligible node, so a single node holding the model is enough. Prepare the same
model on additional nodes when you want any of them to be able to serve it.
@@ -94,8 +95,8 @@ builds and for running the services directly.
## 2. Complete First-Run Setup
On first launch, PAIR opens a setup window that can install available inference
-engines on the local system. PAIR selects Ollama by default when it is available
-for your platform.
+engines on the local system. PAIR preselects Ollama, LM Studio, and llama.cpp when they are
+available for your platform.
1. Review the available engines.
2. Select the engines to install, or skip installation if they are already
@@ -161,6 +162,14 @@ A node is eligible for a request only when it is online, a compatible engine is
running, and the requested model is available there. To test routing across
multiple nodes, prepare the same model on each of those nodes.
+llama.cpp downloads use exact `owner/repository:quantization` IDs. Its Add model
+list is populated from popular GGUF repositories published by `ggml-org`,
+`bartowski`, and `unsloth`. Typing filters those results locally; pressing
+Enter or the search button searches public Hugging Face repositories. PAIR
+offers one verified `Q4_K_M` option per result. Load, Eject, and Delete use that
+same exact model id; Delete removes the selected entry from llama.cpp's
+persistent cache.
+
Refer to [Managing engines](engine-lifecycle.mdx) for details about:
- Installing, starting, stopping, updating, and uninstalling engines.
@@ -192,6 +201,7 @@ engine would normally listen on:
- `11434` for Ollama.
- `1234` for LM Studio.
+- `8080` for llama.cpp.
Tools already pointed at those ports keep working without reconfiguration, and
the engine moves to the next free port.
@@ -220,15 +230,16 @@ The engine therefore has to understand the style you send.
| --- | --- | --- |
| Ollama | Works | Works |
| LM Studio | Works | Not available |
+| llama.cpp | Works | Not available |
-**If you are unsure, send the OpenAI-style request.** It works with either engine,
-and most tools and SDKs use it. Use the Ollama style
+**If you are unsure, send the OpenAI-style request.** It works with every
+supported engine, and most tools and SDKs use it. Use the Ollama style
only when something you already have is written against Ollama's `/api/...` API.
The choice does not affect routing. Both styles are routed across your cluster the
same way, and neither one changes which node serves the request.
-### OpenAI-Style Request (Works With Either Engine)
+### OpenAI-Style Request (Works With Every Engine)
```bash
curl /v1/chat/completions \
@@ -246,9 +257,9 @@ curl /v1/chat/completions \
### Ollama-Style Request (Ollama Endpoint Only)
-Sending this to the LM Studio endpoint fails, because LM Studio does not
-implement Ollama's API. The `-N` flag tells `curl` not to buffer, so it displays
-the response stream as it is generated.
+Sending this to the LM Studio or llama.cpp endpoint fails, because those engines
+do not implement Ollama's API. The `-N` flag tells `curl` not to buffer, so it
+displays the response stream as it is generated.
```bash
curl -N /api/chat \
@@ -292,11 +303,13 @@ free port.
| --- | --- |
| Ollama-compatible proxy | `11434` |
| LM Studio / OpenAI-compatible proxy | `1234` |
+| llama.cpp / OpenAI-compatible proxy | `8080` |
When PAIR takes one of those ports, the engine behind it moves:
- Ollama moves to `11435` or higher.
- LM Studio moves to `1235` or higher.
+- PAIR's managed llama.cpp router runs on `8081`.
**Endpoints** is the authoritative source for the URL to use.
@@ -558,7 +571,7 @@ downloading again.
Updating keeps your settings, logs, cluster identity, and cluster membership, so
a node stays paired across an update. Model weights are untouched as well,
because they belong to the engine rather than to PAIR. Updating PAIR does not
-update Ollama or LM Studio — engines are updated separately from **Engine
+update its inference engines — engines are updated separately from **Engine
settings**, described in [Managing engines](engine-lifecycle.mdx).
You update each machine from that machine. PAIR never updates a peer for you,
diff --git a/docs/inference-dispatcher.mdx b/docs/inference-dispatcher.mdx
index 4462d967..204ee913 100644
--- a/docs/inference-dispatcher.mdx
+++ b/docs/inference-dispatcher.mdx
@@ -5,9 +5,10 @@ SPDX-License-Identifier: Apache-2.0
# Inference dispatcher
-`scripts/inference-dispatcher` is a standalone HTTP client for exercising Ollama
-and LM Studio-compatible servers. It deliberately has no knowledge of Personal
-AI Router, its broker, Electron, JSON-RPC, discovery, or proxy implementation.
+`scripts/inference-dispatcher` is a standalone HTTP client for exercising
+Ollama, LM Studio, and llama.cpp-compatible servers. It deliberately has no
+knowledge of Personal AI Router, its broker, Electron, JSON-RPC, discovery, or
+proxy implementation.
Pointing it at a compatible proxy is indistinguishable from pointing it at a
native model server, which is what makes it useful: the request enters the
cluster router exactly the way a real third-party client's would.
@@ -18,6 +19,7 @@ The executable uses only the Go standard library.
./scripts/inference-dispatcher.sh # one prompt, default Ollama port
./scripts/inference-dispatcher.sh --count 5 --mode parallel # five concurrent prompts
./scripts/inference-dispatcher.sh --backend lmstudio # LM Studio's OpenAI-compatible API
+./scripts/inference-dispatcher.sh --backend llamacpp # llama.cpp's OpenAI-compatible API
./scripts/inference-dispatcher.sh --list-models # live inventory as JSON
./scripts/inference-dispatcher.sh --help
```
@@ -70,9 +72,9 @@ Pass the file with `--config path/to/config.json` or
`INFERENCE_DISPATCHER_*` environment equivalent.
The client always connects to `127.0.0.1`. The default ports are `11434` for
-Ollama and `1234` for LM Studio. Both are the ports the router's proxies claim,
-so requests enter the cluster router rather than a single engine. Use `--port`
-to select another local port.
+Ollama, `1234` for LM Studio, and `8080` for llama.cpp. These are the ports the
+router's proxies claim, so requests enter the cluster router rather than a
+single engine. Use `--port` to select another local port.
## API behavior
@@ -81,6 +83,8 @@ to select another local port.
- LM Studio inventory: `GET /v1/models`, enriched from `GET /api/v1/models` when
that endpoint answers
- LM Studio inference: `POST /v1/chat/completions`
+- llama.cpp inventory: `GET /v1/models`
+- llama.cpp inference: `POST /v1/chat/completions`
`/v1/models` is queried first because the LM Studio proxy answers it by fanning
out across the cluster, while `/api/v1/models` is forwarded to a single node.
diff --git a/docs/overview.mdx b/docs/overview.mdx
index 0efbb5ab..05f4e51c 100644
--- a/docs/overview.mdx
+++ b/docs/overview.mdx
@@ -38,9 +38,9 @@ These terms have specific meanings in PAIR:
is no server, controller, or primary node.
- **Cluster** — the set of nodes you have paired together. A node belongs to at
most one cluster, and it must leave before it can join another.
-- **Engine** — the local inference server that runs models: Ollama or LM Studio.
- PAIR can install, start, stop, and update an engine, or adopt one you already
- run yourself.
+- **Engine** — the local inference server that runs models: Ollama, LM Studio,
+ or llama.cpp. PAIR can install, start, stop, and update these engines; it can
+ also adopt supported existing Ollama and LM Studio installations.
- **Model** — what you prepare on each node. Nodes do not share models, so a
node can serve a request only for a model it already holds. Preparing the same
model on several nodes is what makes those nodes interchangeable.
@@ -108,7 +108,7 @@ Among eligible owners, the proxy applies this precedence:
2. The scheduler's priority order.
3. A deterministic default.
-The scheduler ranks nodes by total pending work across both engines, and each
+The scheduler ranks nodes by total pending work across all engines, and each
proxy also counts requests it has recently dispatched, so a burst of concurrent
requests spreads out instead of waiting for workload reports to catch up.
diff --git a/docs/terminal-interface.mdx b/docs/terminal-interface.mdx
index 1eb99b5f..d33b57ce 100644
--- a/docs/terminal-interface.mdx
+++ b/docs/terminal-interface.mdx
@@ -13,9 +13,8 @@ desktop environment, or over SSH, where the desktop application cannot run. If a
desktop is available, use the desktop application.
**It has known limitations.** It is an operations tool, not a full replacement.
-It cannot list or delete models, change an engine's port, update an engine,
-control engines on other nodes, or show which node served a workload. The full
-list is in
+It cannot delete models, change an engine's port, update an engine, control
+engines on other nodes, or show which node served a workload. The full list is in
[What the Terminal Interface Cannot Do](#what-the-terminal-interface-cannot-do),
and it is worth reading before you depend on it.
@@ -64,9 +63,13 @@ cd
| Flag | Effect |
| --- | --- |
| `--broker-path ` | Use a service binary that is not beside `nvpair-tui` |
+| `--proxy-engines ` | Select the proxy facades to start and show. Defaults to `ollama,lmstudio,llamacpp` |
| `--log-level ` | Verbosity of the terminal interface's own logging: `debug`, `info`, `warn`, or `error`. PAIR also reads this from `NVPAIR_LOG_LEVEL` |
| `--version` | Print the version and exit |
+For example, limit the process to the llama.cpp facade with
+`nvpair-tui --proxy-engines llamacpp`.
+
The interface's own log output goes to stderr, so it never corrupts the display.
Service logs appear on the **Logs** tab instead.
@@ -130,13 +133,14 @@ Tab switching and `q` do not work until you do.
| 1 | **Overview** | Service uptime and version, and an `ok` / `DOWN` table for each worker |
| 2 | **Errors** | Active service errors by severity, age, node, and message |
| 3 | **Nodes** | Nodes discovered on the network, with `Connected` or `In cluster` status |
-| 4 | **Proxies** | Both compatible proxies: listening port, discovered upstreams, and which node is selected |
+| 4 | **Proxies** | Selected compatible facades: listening port, discovered upstreams, and which node is selected |
| 5 | **Workloads** | Live inference workloads: ID, model, engine, state, and age |
| 6 | **Engines** | Local engines: installed, running, healthy, and port |
-| 7 | **Cluster** | This node's identity, cluster membership, and pairing |
-| 8 | **Manual** | Nodes you added by address, with reachability |
-| 9 | **Settings** | Node settings: force ports, cluster auto-sync, and cluster ID and name |
-| 10 | **Logs** | Service log output, with live log-level control |
+| 7 | **Models** | Local models by engine, including loaded/idle state |
+| 8 | **Cluster** | This node's identity, cluster membership, and pairing |
+| 9 | **Manual** | Nodes you added by address, with reachability |
+| 10 | **Settings** | Node settings: force ports, cluster auto-sync, and cluster ID and name |
+| 11 | **Logs** | Service log output, with live log-level control |
## Pair This Machine with Another
@@ -151,7 +155,7 @@ Pairing is the same six-digit PIN exchange the desktop application uses.
This is the easier path, because there is no address to type. Prefer it whenever
PAIR has already discovered the machine you want.
-**To invite a machine by address**, from the **Cluster** tab (7):
+**To invite a machine by address**, from the **Cluster** tab (8):
1. Press `i`.
2. Type the other machine's host, or `host:port` if it is not on the default
@@ -199,7 +203,13 @@ From the **Engines** tab (6), select an engine with `j` / `k`, then:
| `p` | Download a model |
Pressing `p` opens a prompt. Type the model name, for example `qwen4:12b`, and
-press `enter`. Progress appears on the status line.
+press `enter`. llama.cpp names include the repository and quantization, for
+example `ggml-org/gemma-3-1b-it-GGUF:Q4_K_M`. Progress appears on the status
+line.
+
+The **Models** tab (7) lists every running engine's local inventory. Select a
+model and press `enter` to load it into memory or `u` to unload it. State changes
+come from the engine manager and appear as `loaded`, `idle`, or `unknown`.
A node can serve a request only when it is online, a compatible engine is
running, and the requested model is present on that node. To route across several
@@ -214,8 +224,8 @@ engine, and state.
The **Proxies** tab (4) shows each proxy's listening port and whether it is
routing automatically (`selected=auto`) or pinned to one node. Press `g` to
-switch between the two engines, `enter` to pin the highlighted upstream, and `a`
-to return to automatic routing. Leave it on automatic unless you are
+switch between the selected engines, `enter` to pin the highlighted upstream,
+and `a` to return to automatic routing. Leave it on automatic unless you are
deliberately testing one node.
The **Overview** tab (1) reports whether each worker is up. Worker status is
@@ -225,13 +235,13 @@ best-effort. `DOWN` means a worker reported a crash.
On **Errors** (2), press `c` to clear the selected entry.
-On **Logs** (10), scroll with `j` / `k` and the page keys. Set the log level for
+On **Logs** (11), scroll with `j` / `k` and the page keys. Set the log level for
the whole service fleet with `d` (debug), `i` (info), `w` (warn), or `e` (error).
This is the first place to look when something has not started.
## Change Settings
-On **Settings** (9), move with `j` / `k` and press `enter`. Booleans toggle
+On **Settings** (10), move with `j` / `k` and press `enter`. Booleans toggle
immediately. Text fields open for editing, with `enter` to save and `esc` to
cancel.
@@ -243,8 +253,7 @@ for how PAIR arranges ports.
It is an operations tool, not a full replacement for the desktop application:
-- It cannot list or delete models. You can download one, but the interface shows
- no model inventory.
+- It cannot delete models.
- It cannot change an engine's port. The port column is read-only. Use the
desktop application to change it.
- It cannot update an engine or control engines on other cluster nodes.
diff --git a/scripts/inference-dispatcher/client.go b/scripts/inference-dispatcher/client.go
index a759e973..f7eae1fd 100644
--- a/scripts/inference-dispatcher/client.go
+++ b/scripts/inference-dispatcher/client.go
@@ -111,9 +111,12 @@ func decodeObject(data []byte, target any) error {
func (c *backendClient) listModels(ctx context.Context) ([]RegisteredModel, error) {
var models []RegisteredModel
var err error
- if c.cfg.Backend == "lmstudio" {
+ switch c.cfg.Backend {
+ case "lmstudio":
models, err = c.listLMStudioModels(ctx)
- } else {
+ case "llamacpp":
+ models, err = c.listLlamaCPPModels(ctx)
+ default:
models, err = c.listOllamaModels(ctx)
}
if err != nil {
@@ -128,6 +131,14 @@ func (c *backendClient) listModels(ctx context.Context) ([]RegisteredModel, erro
return models, nil
}
+func (c *backendClient) listLlamaCPPModels(ctx context.Context) ([]RegisteredModel, error) {
+ data, err := c.request(ctx, http.MethodGet, "/v1/models", nil)
+ if err != nil {
+ return nil, fmt.Errorf("query llama.cpp models: %w", err)
+ }
+ return parseOpenAIModels(data)
+}
+
func (c *backendClient) listOllamaModels(ctx context.Context) ([]RegisteredModel, error) {
data, err := c.request(ctx, http.MethodGet, "/api/tags", nil)
if err != nil {
@@ -173,7 +184,7 @@ func (c *backendClient) listLMStudioModels(ctx context.Context) ([]RegisteredMod
openAIData, openAIErr := c.request(ctx, http.MethodGet, "/v1/models", nil)
var models []RegisteredModel
if openAIErr == nil {
- parsed, err := parseLMStudioOpenAIModels(openAIData)
+ parsed, err := parseOpenAIModels(openAIData)
if err != nil {
openAIErr = err
} else if len(parsed) == 0 {
@@ -251,7 +262,7 @@ func parseLMStudioNativeModels(data []byte) ([]RegisteredModel, error) {
return models, nil
}
-func parseLMStudioOpenAIModels(data []byte) ([]RegisteredModel, error) {
+func parseOpenAIModels(data []byte) ([]RegisteredModel, error) {
var response struct {
Data []struct {
ID string `json:"id"`
@@ -341,15 +352,19 @@ func (c *backendClient) resolveModel(ctx context.Context) (string, []RegisteredM
}
func (c *backendClient) inferencePath() string {
- if c.cfg.Backend == "lmstudio" {
+ if c.usesOpenAIProtocol() {
return "/v1/chat/completions"
}
return "/api/generate"
}
+func (c *backendClient) usesOpenAIProtocol() bool {
+ return c.cfg.Backend == "lmstudio" || c.cfg.Backend == "llamacpp"
+}
+
func (c *backendClient) infer(ctx context.Context, model, prompt string) (string, error) {
var payload map[string]any
- if c.cfg.Backend == "lmstudio" {
+ if c.usesOpenAIProtocol() {
payload = map[string]any{
"model": model,
"messages": []map[string]string{{"role": "user", "content": prompt}},
@@ -386,8 +401,8 @@ func (c *backendClient) infer(ctx context.Context, model, prompt string) (string
if err != nil {
return "", err
}
- if c.cfg.Backend == "lmstudio" {
- return parseLMStudioResponse(data)
+ if c.usesOpenAIProtocol() {
+ return parseOpenAIResponse(data)
}
var response struct {
Response string `json:"response"`
@@ -398,7 +413,7 @@ func (c *backendClient) infer(ctx context.Context, model, prompt string) (string
return strings.TrimSpace(response.Response), nil
}
-func parseLMStudioResponse(data []byte) (string, error) {
+func parseOpenAIResponse(data []byte) (string, error) {
var response struct {
Choices []struct {
Message struct {
@@ -411,7 +426,7 @@ func parseLMStudioResponse(data []byte) (string, error) {
return "", err
}
if len(response.Choices) == 0 {
- return "", errors.New("LM Studio response contained no choices")
+ return "", errors.New("OpenAI-compatible response contained no choices")
}
choice := response.Choices[0]
switch content := choice.Message.Content.(type) {
diff --git a/scripts/inference-dispatcher/config.go b/scripts/inference-dispatcher/config.go
index 2598dcbc..892f313c 100644
--- a/scripts/inference-dispatcher/config.go
+++ b/scripts/inference-dispatcher/config.go
@@ -23,6 +23,8 @@ const (
// PAIR's managed LM Studio backend is moved behind it starting at 1235;
// pass --port explicitly to reach that directly, which bypasses routing.
defaultLMStudioPort = 1234
+ // llama.cpp's OpenAI-compatible proxy facade uses its stock router port.
+ defaultLlamaCPPPort = 8080
defaultErrorLog = "inference_errors.txt"
maxPromptsPerBatch = 100
)
@@ -289,7 +291,7 @@ func parseConfig(args []string, stderr io.Writer) (Config, error) {
fs.SetOutput(stderr)
var parsedConfigPath string
fs.StringVar(&parsedConfigPath, "config", configPath, "JSON configuration file")
- fs.StringVar(&cfg.Backend, "backend", cfg.Backend, "backend: ollama or lmstudio")
+ fs.StringVar(&cfg.Backend, "backend", cfg.Backend, "backend: ollama, lmstudio, or llamacpp")
fs.StringVar(&cfg.Backend, "provider", cfg.Backend, "alias for --backend")
fs.IntVar(&cfg.Port, "port", cfg.Port, "server port (backend default when omitted)")
fs.StringVar(&cfg.Model, "model", cfg.Model, "model name; omitted or auto selects an available model")
@@ -356,8 +358,8 @@ func parseConfig(args []string, stderr io.Writer) (Config, error) {
}
func validateConfig(cfg Config) error {
- if cfg.Backend != "ollama" && cfg.Backend != "lmstudio" {
- return errors.New("--backend must be ollama or lmstudio")
+ if cfg.Backend != "ollama" && cfg.Backend != "lmstudio" && cfg.Backend != "llamacpp" {
+ return errors.New("--backend must be ollama, lmstudio, or llamacpp")
}
if cfg.Port < 0 || cfg.Port > 65535 {
return errors.New("--port must be between 1 and 65535")
@@ -414,8 +416,12 @@ func effectivePort(cfg Config) int {
if cfg.Port != 0 {
return cfg.Port
}
- if cfg.Backend == "lmstudio" {
+ switch cfg.Backend {
+ case "lmstudio":
return defaultLMStudioPort
+ case "llamacpp":
+ return defaultLlamaCPPPort
+ default:
+ return defaultOllamaPort
}
- return defaultOllamaPort
}
diff --git a/scripts/inference-dispatcher/dispatcher_test.go b/scripts/inference-dispatcher/dispatcher_test.go
index c858fcc7..60ae4124 100644
--- a/scripts/inference-dispatcher/dispatcher_test.go
+++ b/scripts/inference-dispatcher/dispatcher_test.go
@@ -152,6 +152,64 @@ func TestLMStudioFallsBackToOpenAIInventory(t *testing.T) {
}
}
+func TestLlamaCPPUsesOpenAIInventoryAndChat(t *testing.T) {
+ type observedRequest struct {
+ method string
+ path string
+ model string
+ messageCount int
+ }
+ observed := make(chan observedRequest, 2)
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ switch {
+ case r.Method == http.MethodGet && r.URL.Path == "/v1/models":
+ observed <- observedRequest{method: r.Method, path: r.URL.Path}
+ _, _ = w.Write([]byte(`{"data":[{"id":"llama-demo"}]}`))
+ case r.Method == http.MethodPost && r.URL.Path == "/v1/chat/completions":
+ var request struct {
+ Model string `json:"model"`
+ Messages []struct {
+ Role string `json:"role"`
+ Content string `json:"content"`
+ } `json:"messages"`
+ }
+ if err := json.NewDecoder(r.Body).Decode(&request); err != nil {
+ t.Errorf("decode chat request: %v", err)
+ }
+ observed <- observedRequest{
+ method: r.Method,
+ path: r.URL.Path,
+ model: request.Model,
+ messageCount: len(request.Messages),
+ }
+ _, _ = w.Write([]byte(`{"choices":[{"message":{"content":"done"}}]}`))
+ default:
+ http.NotFound(w, r)
+ }
+ }))
+ defer server.Close()
+
+ var stdout, stderr bytes.Buffer
+ exit := runAgainstServer(
+ t,
+ context.Background(),
+ []string{"--backend", "llamacpp", "--prompt", "test"},
+ server.URL,
+ &stdout,
+ &stderr,
+ )
+ if exit != 0 {
+ t.Fatalf("exit=%d stderr=%s", exit, stderr.String())
+ }
+ if got := <-observed; got.method != http.MethodGet || got.path != "/v1/models" {
+ t.Fatalf("inventory request = %+v, want GET /v1/models", got)
+ }
+ if got := <-observed; got.method != http.MethodPost || got.path != "/v1/chat/completions" ||
+ got.model != "llama-demo" || got.messageCount != 1 {
+ t.Fatalf("inference request = %+v, want OpenAI chat for llama-demo", got)
+ }
+}
+
// A Personal AI Router proxy answers /v1/models with the whole cluster's
// inventory and forwards /api/v1/models to one node, so the aggregated list must
// win. The native list still supplies the type and capability fields the
@@ -334,6 +392,17 @@ func TestLMStudioDefaultPort(t *testing.T) {
}
}
+func TestLlamaCPPDefaultPort(t *testing.T) {
+ var stderr bytes.Buffer
+ cfg, err := parseConfig([]string{"--backend", "llamacpp"}, &stderr)
+ if err != nil {
+ t.Fatalf("parse config: %v", err)
+ }
+ if port := effectivePort(cfg); port != 8080 {
+ t.Fatalf("llama.cpp default port=%d, want 8080", port)
+ }
+}
+
func TestResponseTextNeverReachesStdout(t *testing.T) {
const secret = "the capital of France is Paris"
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
diff --git a/scripts/stage-vc-redist.mjs b/scripts/stage-vc-redist.mjs
new file mode 100644
index 00000000..d7729976
--- /dev/null
+++ b/scripts/stage-vc-redist.mjs
@@ -0,0 +1,285 @@
+#!/usr/bin/env node
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+/**
+ * Download and verify the latest Microsoft Visual C++ v14 Redistributable used
+ * by the Windows installers.
+ *
+ * The source URL intentionally follows Microsoft's serviced "latest" package.
+ * Authenticode establishes the publisher and integrity, while the minimum
+ * version rejects a validly signed rollback. The observed version and SHA-256
+ * are recorded for release provenance rather than pinned as build inputs.
+ *
+ * On Windows the operating system reads the signature. Everywhere else it is
+ * read out of the PE directly, because `build:electron:win:*` cross-builds the
+ * Windows installers from Linux and macOS, where there is no PowerShell to ask.
+ * Both paths report the same fields and meet the same policy below; see
+ * scripts/windows-pe-signature.mjs for what the direct read does and does not
+ * establish.
+ */
+
+import { spawnSync } from 'node:child_process'
+import { createHash } from 'node:crypto'
+import { mkdirSync, readFileSync, renameSync, rmSync, writeFileSync } from 'node:fs'
+import { dirname, join, resolve } from 'node:path'
+import { fileURLToPath, pathToFileURL } from 'node:url'
+
+import { readPeSignatureMetadata } from './windows-pe-signature.mjs'
+
+export const VC_REDIST_SOURCE_URL = 'https://aka.ms/vc14/vc_redist.x64.exe'
+export const VC_REDIST_MINIMUM_VERSION = '14.51.36247.0'
+
+const REPO_ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..')
+export const VC_REDIST_OUTPUT_DIR = join(REPO_ROOT, '.build', 'vc-redist')
+export const VC_REDIST_OUTPUT_PATH = join(VC_REDIST_OUTPUT_DIR, 'VC_redist.x64.exe')
+export const VC_REDIST_PROVENANCE_PATH = join(VC_REDIST_OUTPUT_DIR, 'manifest.json')
+
+const MAX_DOWNLOAD_BYTES = 100 * 1024 * 1024
+const DOWNLOAD_TIMEOUT_MS = 10 * 60 * 1000
+
+function versionParts(version) {
+ if (typeof version !== 'string' || !/^\d+(?:\.\d+){2,3}$/.test(version)) {
+ throw new Error(`Invalid Visual C++ Redistributable version "${String(version)}".`)
+ }
+ const parts = version.split('.').map(part => Number.parseInt(part, 10))
+ while (parts.length < 4) parts.push(0)
+ return parts
+}
+
+export function compareVersions(left, right) {
+ const leftParts = versionParts(left)
+ const rightParts = versionParts(right)
+ for (let index = 0; index < leftParts.length; index += 1) {
+ if (leftParts[index] < rightParts[index]) return -1
+ if (leftParts[index] > rightParts[index]) return 1
+ }
+ return 0
+}
+
+function normalizedVersion(metadata) {
+ const candidates = [metadata.productVersion, metadata.fileVersion]
+ for (const candidate of candidates) {
+ if (typeof candidate !== 'string') continue
+ const match = candidate.match(/\d+(?:\.\d+){2,3}/)
+ if (match) return match[0]
+ }
+ throw new Error('The Microsoft-signed package did not report a numeric product version.')
+}
+
+function isMicrosoftSigner(subject) {
+ if (typeof subject !== 'string') return false
+ const distinguishedNames = subject.split(',').map(part => part.trim())
+ return (
+ distinguishedNames.includes('CN=Microsoft Corporation') &&
+ distinguishedNames.includes('O=Microsoft Corporation')
+ )
+}
+
+export function validateAuthenticodeMetadata(metadata) {
+ if (typeof metadata !== 'object' || metadata === null || Array.isArray(metadata)) {
+ throw new Error('The Authenticode probe returned invalid metadata.')
+ }
+ if (metadata.status !== 'Valid') {
+ throw new Error(
+ `Visual C++ Redistributable Authenticode status is "${String(metadata.status)}", not "Valid".`
+ )
+ }
+ if (!isMicrosoftSigner(metadata.signerSubject)) {
+ throw new Error(
+ `Visual C++ Redistributable signer is not Microsoft Corporation: "${String(metadata.signerSubject)}".`
+ )
+ }
+ if (
+ typeof metadata.signerThumbprint !== 'string' ||
+ !/^[0-9a-f]{40}$/i.test(metadata.signerThumbprint)
+ ) {
+ throw new Error('Visual C++ Redistributable signer thumbprint is missing or invalid.')
+ }
+
+ const version = normalizedVersion(metadata)
+ if (compareVersions(version, VC_REDIST_MINIMUM_VERSION) < 0) {
+ throw new Error(
+ `Visual C++ Redistributable ${version} is older than the required ` +
+ `${VC_REDIST_MINIMUM_VERSION}.`
+ )
+ }
+
+ return {
+ version,
+ signerSubject: metadata.signerSubject,
+ signerThumbprint: metadata.signerThumbprint.toUpperCase()
+ }
+}
+
+/**
+ * The environment for the Windows PowerShell child. A `PSModulePath` inherited
+ * from PowerShell 7 points Windows PowerShell at 7's incompatible
+ * Microsoft.PowerShell.Security, so `Get-AuthenticodeSignature` cannot load
+ * (PowerShell/PowerShell#18530). With the variable unset, Windows PowerShell
+ * builds its own default. Windows matches variable names case-insensitively.
+ */
+export function windowsPowerShellEnv(parentEnv, filePath) {
+ const env = Object.fromEntries(
+ Object.entries(parentEnv).filter(([name]) => name.toLowerCase() !== 'psmodulepath')
+ )
+ env.NVPAIR_VC_REDIST_PATH = filePath
+ return env
+}
+
+function windowsAuthenticodeMetadata(filePath) {
+ const command = [
+ "$ErrorActionPreference = 'Stop'",
+ '$signature = Get-AuthenticodeSignature -LiteralPath $env:NVPAIR_VC_REDIST_PATH',
+ '$version = (Get-Item -LiteralPath $env:NVPAIR_VC_REDIST_PATH).VersionInfo',
+ '[ordered]@{',
+ 'status = [string]$signature.Status',
+ "signerSubject = $(if ($null -eq $signature.SignerCertificate) { '' } else { [string]$signature.SignerCertificate.Subject })",
+ "signerThumbprint = $(if ($null -eq $signature.SignerCertificate) { '' } else { [string]$signature.SignerCertificate.Thumbprint })",
+ 'fileVersion = [string]$version.FileVersion',
+ 'productVersion = [string]$version.ProductVersion',
+ '} | ConvertTo-Json -Compress'
+ ].join('\n')
+
+ const result = spawnSync(
+ 'powershell.exe',
+ ['-NoProfile', '-NonInteractive', '-Command', command],
+ {
+ encoding: 'utf8',
+ env: windowsPowerShellEnv(process.env, filePath),
+ windowsHide: true
+ }
+ )
+ if (result.error) {
+ throw new Error(
+ `Unable to run PowerShell Authenticode verification: ${result.error.message}`
+ )
+ }
+ if (result.status !== 0) {
+ throw new Error(
+ `PowerShell Authenticode verification failed: ${result.stderr.trim() || `exit ${String(result.status)}`}`
+ )
+ }
+
+ try {
+ return JSON.parse(result.stdout.trim())
+ } catch (error) {
+ const message = error instanceof Error ? error.message : String(error)
+ throw new Error(`Unable to parse Authenticode metadata: ${message}`)
+ }
+}
+
+function inspectAuthenticode(filePath) {
+ return validateAuthenticodeMetadata(
+ process.platform === 'win32'
+ ? windowsAuthenticodeMetadata(filePath)
+ : readPeSignatureMetadata(filePath)
+ )
+}
+
+function sha256(filePath) {
+ return createHash('sha256').update(readFileSync(filePath)).digest('hex')
+}
+
+function validateResolvedUrl(rawUrl) {
+ const url = new URL(rawUrl)
+ if (url.protocol !== 'https:' || url.hostname !== 'download.visualstudio.microsoft.com') {
+ throw new Error(`Microsoft Redistributable resolved to an unexpected URL: ${rawUrl}`)
+ }
+}
+
+export function createProvenance(resolvedUrl, signature, digest) {
+ validateResolvedUrl(resolvedUrl)
+ if (typeof digest !== 'string' || !/^[0-9a-f]{64}$/.test(digest)) {
+ throw new Error('Visual C++ Redistributable SHA-256 is missing or invalid.')
+ }
+ return {
+ schemaVersion: 1,
+ sourceUrl: VC_REDIST_SOURCE_URL,
+ resolvedUrl,
+ minimumVersion: VC_REDIST_MINIMUM_VERSION,
+ version: signature.version,
+ sha256: digest,
+ signerSubject: signature.signerSubject,
+ signerThumbprint: signature.signerThumbprint
+ }
+}
+
+async function downloadToFile(destination) {
+ const response = await fetch(VC_REDIST_SOURCE_URL, {
+ redirect: 'follow',
+ signal: AbortSignal.timeout(DOWNLOAD_TIMEOUT_MS)
+ })
+ if (!response.ok) {
+ throw new Error(
+ `Download ${VC_REDIST_SOURCE_URL} failed with HTTP ${String(response.status)}.`
+ )
+ }
+ validateResolvedUrl(response.url)
+
+ const declaredLength = Number.parseInt(response.headers.get('content-length') ?? '0', 10)
+ if (declaredLength > MAX_DOWNLOAD_BYTES) {
+ throw new Error(
+ `Visual C++ Redistributable declares ${String(declaredLength)} bytes, exceeding the ` +
+ `${String(MAX_DOWNLOAD_BYTES)}-byte limit.`
+ )
+ }
+
+ const bytes = new Uint8Array(await response.arrayBuffer())
+ if (bytes.byteLength === 0 || bytes.byteLength > MAX_DOWNLOAD_BYTES) {
+ throw new Error(
+ `Visual C++ Redistributable download size ${String(bytes.byteLength)} is invalid.`
+ )
+ }
+ writeFileSync(destination, bytes, { flag: 'wx' })
+ return response.url
+}
+
+function writeProvenance(provenance) {
+ const temporaryPath = `${VC_REDIST_PROVENANCE_PATH}.${String(process.pid)}.tmp`
+ try {
+ writeFileSync(temporaryPath, `${JSON.stringify(provenance, null, 2)}\n`, {
+ encoding: 'utf8',
+ flag: 'wx'
+ })
+ rmSync(VC_REDIST_PROVENANCE_PATH, { force: true })
+ renameSync(temporaryPath, VC_REDIST_PROVENANCE_PATH)
+ } finally {
+ rmSync(temporaryPath, { force: true })
+ }
+}
+
+export async function stageVcRedist() {
+ mkdirSync(VC_REDIST_OUTPUT_DIR, { recursive: true })
+ const temporaryPath = join(
+ VC_REDIST_OUTPUT_DIR,
+ `VC_redist.x64.${String(process.pid)}.download`
+ )
+ rmSync(temporaryPath, { force: true })
+
+ try {
+ console.log(`[vc-redist] downloading ${VC_REDIST_SOURCE_URL}`)
+ const resolvedUrl = await downloadToFile(temporaryPath)
+ const signature = inspectAuthenticode(temporaryPath)
+ const digest = sha256(temporaryPath)
+ const provenance = createProvenance(resolvedUrl, signature, digest)
+
+ rmSync(VC_REDIST_OUTPUT_PATH, { force: true })
+ renameSync(temporaryPath, VC_REDIST_OUTPUT_PATH)
+ writeProvenance(provenance)
+ console.log(
+ `[vc-redist] staged ${signature.version} (${digest}) at ${VC_REDIST_OUTPUT_PATH}`
+ )
+ return provenance
+ } finally {
+ rmSync(temporaryPath, { force: true })
+ }
+}
+
+const invokedPath = process.argv[1]
+if (invokedPath && import.meta.url === pathToFileURL(resolve(invokedPath)).href) {
+ stageVcRedist().catch(error => {
+ console.error(error instanceof Error ? error.message : String(error))
+ process.exitCode = 1
+ })
+}
diff --git a/scripts/stage-vc-redist.test.mjs b/scripts/stage-vc-redist.test.mjs
new file mode 100644
index 00000000..5961c87a
--- /dev/null
+++ b/scripts/stage-vc-redist.test.mjs
@@ -0,0 +1,105 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import assert from 'node:assert/strict'
+import test from 'node:test'
+
+import {
+ compareVersions,
+ createProvenance,
+ validateAuthenticodeMetadata,
+ VC_REDIST_MINIMUM_VERSION,
+ windowsPowerShellEnv
+} from './stage-vc-redist.mjs'
+
+const MICROSOFT_SIGNATURE = {
+ status: 'Valid',
+ signerSubject:
+ 'CN=Microsoft Corporation, O=Microsoft Corporation, L=Redmond, S=Washington, C=US',
+ signerThumbprint: '0123456789abcdef0123456789abcdef01234567',
+ fileVersion: VC_REDIST_MINIMUM_VERSION,
+ productVersion: VC_REDIST_MINIMUM_VERSION
+}
+
+test('compareVersions orders numeric runtime versions', () => {
+ assert.equal(compareVersions('14.50.35719.0', '14.50.35719.0'), 0)
+ assert.equal(compareVersions('14.51.1.0', '14.50.99999.0'), 1)
+ assert.equal(compareVersions('14.9.99999.0', '14.50.1.0'), -1)
+})
+
+test('validateAuthenticodeMetadata accepts Microsoft at the version floor', () => {
+ assert.deepEqual(validateAuthenticodeMetadata(MICROSOFT_SIGNATURE), {
+ version: VC_REDIST_MINIMUM_VERSION,
+ signerSubject: MICROSOFT_SIGNATURE.signerSubject,
+ signerThumbprint: MICROSOFT_SIGNATURE.signerThumbprint.toUpperCase()
+ })
+})
+
+test('validateAuthenticodeMetadata rejects an invalid signature', () => {
+ assert.throws(
+ () => validateAuthenticodeMetadata({ ...MICROSOFT_SIGNATURE, status: 'HashMismatch' }),
+ /Authenticode status is "HashMismatch"/
+ )
+})
+
+test('validateAuthenticodeMetadata rejects a non-Microsoft signer', () => {
+ assert.throws(
+ () =>
+ validateAuthenticodeMetadata({
+ ...MICROSOFT_SIGNATURE,
+ signerSubject: 'CN=Example Corporation, O=Example Corporation, C=US'
+ }),
+ /signer is not Microsoft Corporation/
+ )
+})
+
+test('validateAuthenticodeMetadata rejects a signed rollback', () => {
+ assert.throws(
+ () =>
+ validateAuthenticodeMetadata({
+ ...MICROSOFT_SIGNATURE,
+ fileVersion: '14.49.99999.0',
+ productVersion: '14.49.99999.0'
+ }),
+ /is older than the required/
+ )
+})
+
+test('windowsPowerShellEnv drops a PSModulePath inherited from PowerShell 7', () => {
+ const parentEnv = {
+ Path: 'C:\\Windows\\system32',
+ PSModulePath: 'C:\\Program Files\\PowerShell\\7\\Modules'
+ }
+
+ assert.deepEqual(windowsPowerShellEnv(parentEnv, 'C:\\stage\\VC_redist.x64.exe'), {
+ Path: 'C:\\Windows\\system32',
+ NVPAIR_VC_REDIST_PATH: 'C:\\stage\\VC_redist.x64.exe'
+ })
+ assert.equal(parentEnv.PSModulePath, 'C:\\Program Files\\PowerShell\\7\\Modules')
+})
+
+test('windowsPowerShellEnv drops PSModulePath whatever its case', () => {
+ const parentEnv = { PSMODULEPATH: 'C:\\Program Files\\PowerShell\\7\\Modules' }
+
+ assert.deepEqual(windowsPowerShellEnv(parentEnv, 'C:\\stage\\VC_redist.x64.exe'), {
+ NVPAIR_VC_REDIST_PATH: 'C:\\stage\\VC_redist.x64.exe'
+ })
+})
+
+test('createProvenance records the verified package identity', () => {
+ const resolvedUrl =
+ 'https://download.visualstudio.microsoft.com/download/pr/package/VC_redist.x64.exe'
+ const sha256 = 'a'.repeat(64)
+ const signature = validateAuthenticodeMetadata(MICROSOFT_SIGNATURE)
+
+ assert.deepEqual(createProvenance(resolvedUrl, signature, sha256), {
+ schemaVersion: 1,
+ sourceUrl: 'https://aka.ms/vc14/vc_redist.x64.exe',
+ resolvedUrl,
+ minimumVersion: VC_REDIST_MINIMUM_VERSION,
+ version: VC_REDIST_MINIMUM_VERSION,
+ sha256,
+ signerSubject: MICROSOFT_SIGNATURE.signerSubject,
+ signerThumbprint: MICROSOFT_SIGNATURE.signerThumbprint.toUpperCase()
+ })
+})
diff --git a/scripts/windows-pe-signature.mjs b/scripts/windows-pe-signature.mjs
new file mode 100644
index 00000000..b3a4582b
--- /dev/null
+++ b/scripts/windows-pe-signature.mjs
@@ -0,0 +1,460 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+/**
+ * Read an Authenticode signature out of a Windows PE file on any platform.
+ *
+ * The Windows installers cross-build on Linux, where `Get-AuthenticodeSignature`
+ * does not exist, so the staged Visual C++ Redistributable has to be inspected
+ * without the operating system's help. This module reads the PE's certificate
+ * table directly and reports the same fields the PowerShell probe does.
+ *
+ * What it establishes: the file carries a PKCS#7 signature; the digest that was
+ * signed is the digest of these bytes; the signing certificate chains, by
+ * issuance and by signature, to the Microsoft certificate authority pinned
+ * below; and the version the PE reports about itself. What it does not
+ * establish: certificate validity windows, revocation, or the RFC 3161
+ * countersignature. A signing certificate that has expired since it signed is
+ * normal and still accepted here, which is the main reason this is not a
+ * general-purpose Authenticode verifier — on Windows the operating system
+ * remains the one making that judgement.
+ */
+
+import { X509Certificate, createHash } from 'node:crypto'
+import { readFileSync } from 'node:fs'
+
+// Authenticode stores a PKCS#7 SignedData whose encapsulated content is an
+// SpcIndirectDataContent carrying the digest of the image.
+const SIGNED_DATA_OID = '1.2.840.113549.1.7.2'
+const SPC_INDIRECT_DATA_OID = '1.3.6.1.4.1.311.2.1.4'
+
+const DIGEST_ALGORITHM_OIDS = new Map([
+ ['1.3.14.3.2.26', 'sha1'],
+ ['2.16.840.1.101.3.4.2.1', 'sha256'],
+ ['2.16.840.1.101.3.4.2.2', 'sha384'],
+ ['2.16.840.1.101.3.4.2.3', 'sha512']
+])
+
+// The certificate authority the embedded chain is required to terminate in.
+//
+// A PE embeds the signing certificate and its issuers but not the root, so this
+// pins the issuing CA rather than a root. Pinning is what makes the signer
+// subject meaningful: without an anchor, any self-signed certificate could name
+// itself Microsoft Corporation. When Microsoft moves to a new CA the staging
+// fails with the new fingerprint in the message, and that fingerprint is added
+// here after being confirmed against a Microsoft-published certificate.
+export const MICROSOFT_ISSUER_FINGERPRINTS = new Set([
+ // CN=Microsoft Code Signing PCA 2024, O=Microsoft Corporation, C=US — the
+ // last certificate in the chain embedded in VC_redist.x64.exe 14.51.36247.0.
+ '3D:AD:FA:F8:12:DD:1B:BA:EF:45:83:4C:CB:D1:88:F3:CD:97:13:9E:2E:D1:AC:A6:9C:2D:D6:30:82:14:2F:8F'
+])
+
+const PE32_MAGIC = 0x10b
+const PE32PLUS_MAGIC = 0x20b
+// Offsets inside the optional header. The checksum sits at the same place in
+// both variants; the data directory array does not.
+const CHECKSUM_OFFSET_IN_OPTIONAL_HEADER = 64
+const DATA_DIRECTORY_OFFSET_IN_OPTIONAL_HEADER = new Map([
+ [PE32_MAGIC, 96],
+ [PE32PLUS_MAGIC, 112]
+])
+const DATA_DIRECTORY_ENTRY_SIZE = 8
+const RESOURCE_DIRECTORY_INDEX = 2
+const CERTIFICATE_TABLE_INDEX = 4
+
+const WIN_CERTIFICATE_HEADER_SIZE = 8
+const WIN_CERTIFICATE_TYPE_PKCS_SIGNED_DATA = 0x0002
+
+const RESOURCE_TYPE_VERSION = 16
+const VS_FIXEDFILEINFO_SIGNATURE = 0xfeef04bd
+
+export function readTypeLengthValue(buffer, offset) {
+ if (offset + 2 > buffer.length) {
+ throw new Error('The signature is truncated: a DER header runs past the end.')
+ }
+ const tag = buffer[offset]
+ const firstLengthByte = buffer[offset + 1]
+ let length = firstLengthByte
+ let headerLength = 2
+
+ if ((firstLengthByte & 0x80) !== 0) {
+ const lengthByteCount = firstLengthByte & 0x7f
+ // Indefinite length is not legal in DER, and four bytes already covers
+ // any signature that fits the download size limit.
+ if (lengthByteCount === 0 || lengthByteCount > 4) {
+ throw new Error(
+ `The signature uses an unsupported DER length of ${lengthByteCount} bytes.`
+ )
+ }
+ length = 0
+ for (let index = 0; index < lengthByteCount; index += 1) {
+ length = length * 256 + buffer[offset + 2 + index]
+ }
+ headerLength = 2 + lengthByteCount
+ }
+
+ const start = offset + headerLength
+ const end = start + length
+ if (end > buffer.length) {
+ throw new Error('The signature is truncated: a DER value runs past the end.')
+ }
+ // `offset` is the element itself, `start` only its content: a certificate
+ // has to be handed on with its own header intact.
+ return { tag, offset, start, end }
+}
+
+export function derChildren(buffer, node) {
+ const nodes = []
+ let offset = node.start
+ while (offset < node.end) {
+ const child = readTypeLengthValue(buffer, offset)
+ nodes.push(child)
+ offset = child.end
+ }
+ return nodes
+}
+
+function expectTag(node, tag, description) {
+ if (node.tag !== tag) {
+ throw new Error(
+ `The signature is malformed: expected ${description} (tag 0x${tag.toString(16)}), ` +
+ `found tag 0x${node.tag.toString(16)}.`
+ )
+ }
+ return node
+}
+
+export function decodeObjectIdentifier(buffer, node) {
+ const bytes = buffer.subarray(node.start, node.end)
+ if (bytes.length === 0) {
+ throw new Error('The signature is malformed: an empty object identifier.')
+ }
+ const parts = [Math.floor(bytes[0] / 40), bytes[0] % 40]
+ let value = 0
+ for (let index = 1; index < bytes.length; index += 1) {
+ value = value * 128 + (bytes[index] & 0x7f)
+ if ((bytes[index] & 0x80) === 0) {
+ parts.push(value)
+ value = 0
+ }
+ }
+ return parts.join('.')
+}
+
+/**
+ * The PE offsets this module needs: what to exclude from the image digest, and
+ * where the certificate table and resources live.
+ */
+function readPortableExecutableLayout(buffer) {
+ if (buffer.length < 64 || buffer[0] !== 0x4d || buffer[1] !== 0x5a) {
+ throw new Error('The file is not a Windows executable (no MZ header).')
+ }
+ const peHeaderOffset = buffer.readUInt32LE(0x3c)
+ if (peHeaderOffset + 24 > buffer.length || buffer.readUInt32LE(peHeaderOffset) !== 0x00004550) {
+ throw new Error('The file is not a Windows executable (no PE signature).')
+ }
+
+ const sectionCount = buffer.readUInt16LE(peHeaderOffset + 6)
+ const optionalHeaderSize = buffer.readUInt16LE(peHeaderOffset + 20)
+ const optionalHeaderOffset = peHeaderOffset + 24
+ const magic = buffer.readUInt16LE(optionalHeaderOffset)
+ const dataDirectoryOffset = DATA_DIRECTORY_OFFSET_IN_OPTIONAL_HEADER.get(magic)
+ if (dataDirectoryOffset === undefined) {
+ throw new Error(
+ `The executable has an unknown optional header magic 0x${magic.toString(16)}.`
+ )
+ }
+
+ const directoryOffset = index =>
+ optionalHeaderOffset + dataDirectoryOffset + index * DATA_DIRECTORY_ENTRY_SIZE
+ const readDirectory = index => ({
+ address: buffer.readUInt32LE(directoryOffset(index)),
+ size: buffer.readUInt32LE(directoryOffset(index) + 4)
+ })
+
+ const sectionTableOffset = optionalHeaderOffset + optionalHeaderSize
+ const sections = []
+ for (let index = 0; index < sectionCount; index += 1) {
+ const offset = sectionTableOffset + index * 40
+ sections.push({
+ virtualAddress: buffer.readUInt32LE(offset + 12),
+ virtualSize: buffer.readUInt32LE(offset + 8),
+ rawOffset: buffer.readUInt32LE(offset + 20),
+ rawSize: buffer.readUInt32LE(offset + 16)
+ })
+ }
+
+ return {
+ checksumOffset: optionalHeaderOffset + CHECKSUM_OFFSET_IN_OPTIONAL_HEADER,
+ certificateDirectoryOffset: directoryOffset(CERTIFICATE_TABLE_INDEX),
+ certificateTable: readDirectory(CERTIFICATE_TABLE_INDEX),
+ resourceDirectory: readDirectory(RESOURCE_DIRECTORY_INDEX),
+ sections
+ }
+}
+
+function fileOffsetForAddress(layout, address) {
+ for (const section of layout.sections) {
+ const end = section.virtualAddress + Math.max(section.virtualSize, section.rawSize)
+ if (address >= section.virtualAddress && address < end) {
+ return section.rawOffset + (address - section.virtualAddress)
+ }
+ }
+ throw new Error(`The executable has no section containing address 0x${address.toString(16)}.`)
+}
+
+/**
+ * The Authenticode image digest: the whole file except the three regions a
+ * signature cannot cover — its own checksum, the certificate table's data
+ * directory entry, and the certificate table itself.
+ */
+function imageDigest(buffer, layout, algorithm) {
+ const certificateTableOffset = layout.certificateTable.address
+ const hash = createHash(algorithm)
+ hash.update(buffer.subarray(0, layout.checksumOffset))
+ hash.update(buffer.subarray(layout.checksumOffset + 4, layout.certificateDirectoryOffset))
+ hash.update(
+ buffer.subarray(
+ layout.certificateDirectoryOffset + DATA_DIRECTORY_ENTRY_SIZE,
+ certificateTableOffset
+ )
+ )
+ // Anything appended after the certificate table is covered, so a trailing
+ // payload cannot be swapped out. For these packages there is none.
+ hash.update(buffer.subarray(certificateTableOffset + layout.certificateTable.size))
+ return hash.digest('hex')
+}
+
+function readCertificateTable(buffer, layout) {
+ const { address, size } = layout.certificateTable
+ if (address === 0 || size === 0) {
+ throw new Error('The executable is not signed: it has no certificate table.')
+ }
+ if (address + size > buffer.length) {
+ throw new Error('The executable declares a certificate table past the end of the file.')
+ }
+ const certificateType = buffer.readUInt16LE(address + 6)
+ if (certificateType !== WIN_CERTIFICATE_TYPE_PKCS_SIGNED_DATA) {
+ throw new Error(
+ `The executable's certificate table holds type ${certificateType}, not PKCS#7 signed data.`
+ )
+ }
+ const declaredLength = buffer.readUInt32LE(address)
+ return buffer.subarray(address + WIN_CERTIFICATE_HEADER_SIZE, address + declaredLength)
+}
+
+/**
+ * Pull the certificates and the signed image digest out of the PKCS#7 blob.
+ */
+function readSignedData(pkcs7) {
+ const contentInfo = expectTag(readTypeLengthValue(pkcs7, 0), 0x30, 'a PKCS#7 ContentInfo')
+ const [contentType, wrappedContent] = derChildren(pkcs7, contentInfo)
+ if (
+ decodeObjectIdentifier(pkcs7, expectTag(contentType, 0x06, 'a content type')) !==
+ SIGNED_DATA_OID
+ ) {
+ throw new Error('The executable is not signed with PKCS#7 SignedData.')
+ }
+
+ const signedData = expectTag(
+ derChildren(pkcs7, expectTag(wrappedContent, 0xa0, 'the SignedData wrapper'))[0],
+ 0x30,
+ 'a SignedData'
+ )
+ const members = derChildren(pkcs7, signedData)
+ // version, digestAlgorithms, contentInfo, then the optional [0] certificates
+ // and [1] CRLs, then signerInfos.
+ const encapsulated = expectTag(members[2], 0x30, 'an encapsulated ContentInfo')
+ const certificateSet = members.find(member => member.tag === 0xa0)
+ if (certificateSet === undefined) {
+ throw new Error('The signature carries no certificates.')
+ }
+
+ const [encapsulatedType, encapsulatedContent] = derChildren(pkcs7, encapsulated)
+ if (
+ decodeObjectIdentifier(pkcs7, expectTag(encapsulatedType, 0x06, 'a content type')) !==
+ SPC_INDIRECT_DATA_OID
+ ) {
+ throw new Error('The signature does not carry an Authenticode image digest.')
+ }
+
+ const indirectData = expectTag(
+ derChildren(pkcs7, expectTag(encapsulatedContent, 0xa0, 'the content wrapper'))[0],
+ 0x30,
+ 'an SpcIndirectDataContent'
+ )
+ const digestInfo = expectTag(derChildren(pkcs7, indirectData)[1], 0x30, 'the signed digest')
+ const [algorithmIdentifier, digestValue] = derChildren(pkcs7, digestInfo)
+ const algorithmOid = decodeObjectIdentifier(
+ pkcs7,
+ expectTag(
+ derChildren(pkcs7, expectTag(algorithmIdentifier, 0x30, 'a digest algorithm'))[0],
+ 0x06,
+ 'a digest algorithm identifier'
+ )
+ )
+ const algorithm = DIGEST_ALGORITHM_OIDS.get(algorithmOid)
+ if (algorithm === undefined) {
+ throw new Error(`The signature uses an unsupported digest algorithm ${algorithmOid}.`)
+ }
+
+ return {
+ algorithm,
+ digest: pkcs7
+ .subarray(
+ expectTag(digestValue, 0x04, 'the signed digest value').start,
+ digestValue.end
+ )
+ .toString('hex'),
+ certificates: derChildren(pkcs7, certificateSet).map(
+ node => new X509Certificate(pkcs7.subarray(node.offset, node.end))
+ )
+ }
+}
+
+/**
+ * Walk the embedded certificates from the signing certificate outwards, and
+ * require the chain to terminate in a pinned Microsoft certificate authority.
+ *
+ * Only issuance and signature are checked, not the validity window: Microsoft's
+ * signing certificates outlive their own signatures by design, and the
+ * countersignature recording when the signing happened is not verified here.
+ */
+export function verifyChainToMicrosoftIssuer(certificates) {
+ const issuesAnother = certificate =>
+ certificates.some(other => other !== certificate && other.checkIssued(certificate))
+ const leaf = certificates.find(certificate => !issuesAnother(certificate))
+ if (leaf === undefined) {
+ throw new Error(
+ 'The signature has no signing certificate: every certificate issues another.'
+ )
+ }
+
+ let current = leaf
+ const seen = new Set([current.fingerprint256])
+ for (;;) {
+ const issuer = certificates.find(
+ candidate => candidate !== current && current.checkIssued(candidate)
+ )
+ if (issuer === undefined) break
+ if (!current.verify(issuer.publicKey)) {
+ throw new Error(
+ `The certificate chain is broken: "${distinguishedName(current.subject)}" is not ` +
+ 'signed by its issuer.'
+ )
+ }
+ if (seen.has(issuer.fingerprint256)) {
+ throw new Error('The certificate chain loops back on itself.')
+ }
+ seen.add(issuer.fingerprint256)
+ current = issuer
+ }
+
+ if (!MICROSOFT_ISSUER_FINGERPRINTS.has(current.fingerprint256)) {
+ throw new Error(
+ `The certificate chain ends at "${distinguishedName(current.subject)}" ` +
+ `(${current.fingerprint256}), which is not a pinned Microsoft ` +
+ 'certificate authority.'
+ )
+ }
+ return leaf
+}
+
+/**
+ * One level of the resource tree: named entries first, then the ones addressed
+ * by numeric id. A directory entry's offset is relative to the tree's root.
+ */
+function resourceEntries(buffer, directoryOffset) {
+ const namedCount = buffer.readUInt16LE(directoryOffset + 12)
+ const idCount = buffer.readUInt16LE(directoryOffset + 14)
+ const entries = []
+ for (let index = 0; index < namedCount + idCount; index += 1) {
+ const offset = directoryOffset + 16 + index * 8
+ const name = buffer.readUInt32LE(offset)
+ const data = buffer.readUInt32LE(offset + 4)
+ entries.push({
+ id: (name & 0x80000000) === 0 ? name : null,
+ isDirectory: (data & 0x80000000) !== 0,
+ offset: data & 0x7fffffff
+ })
+ }
+ return entries
+}
+
+/**
+ * VS_FIXEDFILEINFO sits behind a UTF-16 key and alignment padding, so it is
+ * found by its signature rather than by counting bytes to it.
+ */
+function fixedFileInfoOffset(buffer, versionOffset) {
+ for (let offset = versionOffset; offset < versionOffset + 64; offset += 4) {
+ if (buffer.readUInt32LE(offset) === VS_FIXEDFILEINFO_SIGNATURE) return offset
+ }
+ throw new Error("The executable's version resource carries no VS_FIXEDFILEINFO.")
+}
+
+/**
+ * The version the PE reports about itself, from its first RT_VERSION resource.
+ */
+function readVersionInfo(buffer, layout) {
+ const resourceRoot = fileOffsetForAddress(layout, layout.resourceDirectory.address)
+ const typeEntry = resourceEntries(buffer, resourceRoot).find(
+ entry => entry.id === RESOURCE_TYPE_VERSION && entry.isDirectory
+ )
+ if (typeEntry === undefined) {
+ throw new Error('The executable carries no version resource.')
+ }
+ const nameEntry = resourceEntries(buffer, resourceRoot + typeEntry.offset)[0]
+ const languageEntry = resourceEntries(buffer, resourceRoot + nameEntry.offset)[0]
+ // The leaf addresses an IMAGE_RESOURCE_DATA_ENTRY, and that entry's own
+ // OffsetToData is an image address rather than a resource-relative one.
+ const dataEntryOffset = resourceRoot + languageEntry.offset
+ const versionOffset = fileOffsetForAddress(layout, buffer.readUInt32LE(dataEntryOffset))
+
+ const fixedInfoOffset = fixedFileInfoOffset(buffer, versionOffset)
+ const fileVersion = versionFromParts(
+ buffer.readUInt32LE(fixedInfoOffset + 8),
+ buffer.readUInt32LE(fixedInfoOffset + 12)
+ )
+ const productVersion = versionFromParts(
+ buffer.readUInt32LE(fixedInfoOffset + 16),
+ buffer.readUInt32LE(fixedInfoOffset + 20)
+ )
+ return { fileVersion, productVersion }
+}
+
+export function versionFromParts(most, least) {
+ return [most >>> 16, most & 0xffff, least >>> 16, least & 0xffff].join('.')
+}
+
+/**
+ * Node lists a subject one attribute per line, least significant first, and
+ * spells stateOrProvinceName `ST`. PowerShell prints a single comma-separated
+ * line, most significant first, and spells it `S`. Following PowerShell keeps
+ * the recorded signer identical whichever platform staged the package.
+ */
+export function distinguishedName(subject) {
+ return subject
+ .split('\n')
+ .filter(attribute => attribute.length > 0)
+ .reverse()
+ .map(attribute => attribute.replace(/^ST=/, 'S='))
+ .join(', ')
+}
+
+export function readPeSignatureMetadata(filePath) {
+ const buffer = readFileSync(filePath)
+ const layout = readPortableExecutableLayout(buffer)
+ const { algorithm, digest, certificates } = readSignedData(readCertificateTable(buffer, layout))
+
+ const leaf = verifyChainToMicrosoftIssuer(certificates)
+ const computed = imageDigest(buffer, layout, algorithm)
+ const { fileVersion, productVersion } = readVersionInfo(buffer, layout)
+
+ return {
+ status: computed === digest ? 'Valid' : 'HashMismatch',
+ signerSubject: distinguishedName(leaf.subject),
+ signerThumbprint: leaf.fingerprint.replaceAll(':', ''),
+ fileVersion,
+ productVersion
+ }
+}
diff --git a/scripts/windows-pe-signature.test.mjs b/scripts/windows-pe-signature.test.mjs
new file mode 100644
index 00000000..f27e4665
--- /dev/null
+++ b/scripts/windows-pe-signature.test.mjs
@@ -0,0 +1,197 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import assert from 'node:assert/strict'
+import test from 'node:test'
+
+import {
+ decodeObjectIdentifier,
+ derChildren,
+ distinguishedName,
+ MICROSOFT_ISSUER_FINGERPRINTS,
+ readTypeLengthValue,
+ verifyChainToMicrosoftIssuer,
+ versionFromParts
+} from './windows-pe-signature.mjs'
+
+const DER_SEQUENCE = 0x30
+const DER_OBJECT_IDENTIFIER = 0x06
+
+const objectIdentifier = bytes => Buffer.from([DER_OBJECT_IDENTIFIER, bytes.length, ...bytes])
+
+/**
+ * A certificate stood up from a name and the name of its issuer, exposing only
+ * what the chain walk uses. `checkIssued(candidate)` answers whether this
+ * certificate was issued by the candidate, the same direction Node's
+ * X509Certificate uses.
+ */
+function certificate({ name, fingerprint256, issuedBy = null, signatureValid = true }) {
+ return {
+ subject: `CN=${name}`,
+ fingerprint256,
+ publicKey: { name },
+ checkIssued: candidate => issuedBy !== null && candidate.publicKey.name === issuedBy,
+ verify: publicKey => signatureValid && publicKey.name === issuedBy
+ }
+}
+
+const PINNED_ISSUER_FINGERPRINT = [...MICROSOFT_ISSUER_FINGERPRINTS][0]
+
+test('every pinned issuer is recorded as a SHA-256 fingerprint', () => {
+ // The chain walk compares against `fingerprint256`, so a SHA-1 thumbprint
+ // pasted in here would reject every file instead of failing visibly.
+ assert.ok(MICROSOFT_ISSUER_FINGERPRINTS.size > 0)
+ for (const fingerprint of MICROSOFT_ISSUER_FINGERPRINTS) {
+ assert.match(fingerprint, /^([0-9A-F]{2}:){31}[0-9A-F]{2}$/)
+ }
+})
+
+test('readTypeLengthValue reads a short-form element', () => {
+ const buffer = Buffer.from([DER_SEQUENCE, 0x03, 0x01, 0x02, 0x03])
+ assert.deepEqual(readTypeLengthValue(buffer, 0), {
+ tag: DER_SEQUENCE,
+ offset: 0,
+ start: 2,
+ end: 5
+ })
+})
+
+test('readTypeLengthValue keeps the element offset apart from its content', () => {
+ // A certificate has to be handed to X509Certificate with its own header, so
+ // `offset` addressing the element and `start` addressing the content is the
+ // distinction the signature reader depends on.
+ const buffer = Buffer.concat([
+ Buffer.from([DER_SEQUENCE, 0x81, 0xc8]),
+ Buffer.alloc(0xc8, 0x41)
+ ])
+ const node = readTypeLengthValue(buffer, 0)
+ assert.equal(node.offset, 0)
+ assert.equal(node.start, 3)
+ assert.equal(node.end, buffer.length)
+})
+
+test('readTypeLengthValue reads a two-byte length', () => {
+ const buffer = Buffer.concat([
+ Buffer.from([DER_SEQUENCE, 0x82, 0x01, 0x00]),
+ Buffer.alloc(256, 0x41)
+ ])
+ assert.equal(readTypeLengthValue(buffer, 0).end, 260)
+})
+
+test('readTypeLengthValue rejects a value that runs past the end', () => {
+ assert.throws(
+ () => readTypeLengthValue(Buffer.from([DER_SEQUENCE, 0x10, 0x00]), 0),
+ /truncated/
+ )
+})
+
+test('readTypeLengthValue rejects an unsupported length width', () => {
+ assert.throws(
+ () => readTypeLengthValue(Buffer.from([DER_SEQUENCE, 0x85, 0, 0, 0, 0, 0]), 0),
+ /unsupported DER length of 5 bytes/
+ )
+})
+
+test('derChildren walks the elements of a sequence', () => {
+ const buffer = Buffer.from([DER_SEQUENCE, 0x06, 0x02, 0x01, 0x07, 0x02, 0x02, 0x08, 0x09])
+ const children = derChildren(buffer, readTypeLengthValue(buffer, 0))
+ assert.deepEqual(
+ children.map(child => [child.tag, child.end - child.start]),
+ [
+ [0x02, 1],
+ [0x02, 2]
+ ]
+ )
+})
+
+test('decodeObjectIdentifier decodes the PKCS#7 and Authenticode identifiers', () => {
+ const signedData = objectIdentifier([0x2a, 0x86, 0x48, 0x86, 0xf7, 0x0d, 0x01, 0x07, 0x02])
+ const spcIndirectData = objectIdentifier([
+ 0x2b, 0x06, 0x01, 0x04, 0x01, 0x82, 0x37, 0x02, 0x01, 0x04
+ ])
+ assert.equal(
+ decodeObjectIdentifier(signedData, readTypeLengthValue(signedData, 0)),
+ '1.2.840.113549.1.7.2'
+ )
+ // 311 needs two continuation bytes, so this covers the multi-byte arc.
+ assert.equal(
+ decodeObjectIdentifier(spcIndirectData, readTypeLengthValue(spcIndirectData, 0)),
+ '1.3.6.1.4.1.311.2.1.4'
+ )
+})
+
+test('versionFromParts unpacks a VS_FIXEDFILEINFO version', () => {
+ // 14.51.36247.0, as the staged redistributable reports itself.
+ assert.equal(versionFromParts(0x000e0033, 0x8d970000), '14.51.36247.0')
+ assert.equal(versionFromParts(0, 0), '0.0.0.0')
+ assert.equal(versionFromParts(0xffffffff, 0xffffffff), '65535.65535.65535.65535')
+})
+
+test('distinguishedName reorders a subject the way PowerShell prints it', () => {
+ assert.equal(
+ distinguishedName('C=US\nST=Washington\nL=Redmond\nO=Microsoft Corporation\n'),
+ 'O=Microsoft Corporation, L=Redmond, S=Washington, C=US'
+ )
+})
+
+test('verifyChainToMicrosoftIssuer returns the signing certificate', () => {
+ const issuer = certificate({
+ name: 'Microsoft Code Signing PCA 2024',
+ fingerprint256: PINNED_ISSUER_FINGERPRINT
+ })
+ const leaf = certificate({
+ name: 'Microsoft Corporation',
+ fingerprint256: 'AA:BB',
+ issuedBy: 'Microsoft Code Signing PCA 2024'
+ })
+ // Certificate order in the signature is not specified, so the walk has to
+ // find the signing certificate rather than take the first one.
+ assert.equal(verifyChainToMicrosoftIssuer([issuer, leaf]), leaf)
+})
+
+test('verifyChainToMicrosoftIssuer rejects a chain to an unpinned authority', () => {
+ const issuer = certificate({ name: 'Example Root CA', fingerprint256: 'CC:DD' })
+ const leaf = certificate({
+ name: 'Microsoft Corporation',
+ fingerprint256: 'AA:BB',
+ issuedBy: 'Example Root CA'
+ })
+ assert.throws(
+ () => verifyChainToMicrosoftIssuer([leaf, issuer]),
+ /ends at "CN=Example Root CA" \(CC:DD\), which is not a pinned Microsoft/
+ )
+})
+
+test('verifyChainToMicrosoftIssuer rejects a certificate its issuer did not sign', () => {
+ const issuer = certificate({
+ name: 'Microsoft Code Signing PCA 2024',
+ fingerprint256: PINNED_ISSUER_FINGERPRINT
+ })
+ const leaf = certificate({
+ name: 'Microsoft Corporation',
+ fingerprint256: 'AA:BB',
+ issuedBy: 'Microsoft Code Signing PCA 2024',
+ signatureValid: false
+ })
+ assert.throws(
+ () => verifyChainToMicrosoftIssuer([leaf, issuer]),
+ /chain is broken: "CN=Microsoft Corporation" is not signed by its issuer/
+ )
+})
+
+test('verifyChainToMicrosoftIssuer rejects a chain that never terminates', () => {
+ const first = certificate({ name: 'First CA', fingerprint256: 'CC:DD', issuedBy: 'Second CA' })
+ const second = certificate({ name: 'Second CA', fingerprint256: 'EE:FF', issuedBy: 'First CA' })
+ const leaf = certificate({
+ name: 'Microsoft Corporation',
+ fingerprint256: 'AA:BB',
+ issuedBy: 'First CA'
+ })
+ assert.throws(() => verifyChainToMicrosoftIssuer([leaf, first, second]), /loops back on itself/)
+})
+
+test('verifyChainToMicrosoftIssuer rejects a signature with no signing certificate', () => {
+ const first = certificate({ name: 'First CA', fingerprint256: 'CC:DD', issuedBy: 'Second CA' })
+ const second = certificate({ name: 'Second CA', fingerprint256: 'EE:FF', issuedBy: 'First CA' })
+ assert.throws(() => verifyChainToMicrosoftIssuer([first, second]), /no signing certificate/)
+})
diff --git a/scripts/windows-vc-runtime-installers.test.mjs b/scripts/windows-vc-runtime-installers.test.mjs
new file mode 100644
index 00000000..da96e6a7
--- /dev/null
+++ b/scripts/windows-vc-runtime-installers.test.mjs
@@ -0,0 +1,66 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import assert from 'node:assert/strict'
+import { readFileSync } from 'node:fs'
+import { dirname, resolve } from 'node:path'
+import test from 'node:test'
+import { fileURLToPath } from 'node:url'
+
+const REPO_ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..')
+
+function repositoryFile(relativePath) {
+ return readFileSync(resolve(REPO_ROOT, relativePath), 'utf8')
+}
+
+function assertRuntimeExitHandling(source, exitCodeVariable) {
+ assert.match(source, /\/install \/quiet \/norestart/)
+ for (const successCode of [0, 1638, 1641, 3010]) {
+ assert.ok(
+ source.includes(`${exitCodeVariable} == ${String(successCode)}`),
+ `missing accepted runtime exit code ${String(successCode)}`
+ )
+ }
+ assert.match(source, /SetRebootFlag true/)
+ assert.match(source, /SetErrorLevel 3/)
+}
+
+test('both desktop Windows architectures stage the shared runtime package', () => {
+ const packageJson = JSON.parse(repositoryFile('desktop/package.json'))
+
+ assert.match(packageJson.scripts['build:electron:win:x64'], /^npm run stage:vc-redist &&/)
+ assert.match(packageJson.scripts['build:electron:win:arm64'], /^npm run stage:vc-redist &&/)
+})
+
+test('desktop packaging verifies and installs the staged runtime', () => {
+ const builderConfig = repositoryFile('desktop/electron-builder.config.ts')
+ const installer = repositoryFile('desktop/scripts/build/installer.nsh')
+
+ assert.match(builderConfig, /function assertVcRedistPackagingInput\(\): void/)
+ assert.match(builderConfig, /createHash\('sha256'\)/)
+ assert.match(builderConfig, /from: vcRedistStagedPath/)
+ assert.match(builderConfig, /to: 'installer-tools\/VC_redist\.x64\.exe'/)
+ assert.match(builderConfig, /signExts: \['!VC_redist\.x64\.exe'\]/)
+ assert.match(
+ installer,
+ /!macro customInstall\s+!insertmacro pairAssertPayloadInstalled\s+!insertmacro pairInstallVcRuntime\s+!insertmacro pairAddFirewallRules\s+!macroend/
+ )
+ assert.match(installer, /Delete "\$INSTDIR\\resources\\installer-tools\\VC_redist\.x64\.exe"/)
+ assertRuntimeExitHandling(installer, '$8')
+})
+
+test('standalone services installer stages and installs the shared runtime', () => {
+ const buildScript = repositoryFile('services/installer_build.bat')
+ const installer = repositoryFile('services/installer/nvpair-setup.nsi')
+
+ const stageIndex = buildScript.indexOf('node "%ROOT%..\\scripts\\stage-vc-redist.mjs"')
+ const makensisIndex = buildScript.indexOf('"%MAKENSIS%" /V3')
+ assert.ok(stageIndex >= 0, 'services installer build does not stage the runtime')
+ assert.ok(stageIndex < makensisIndex, 'services installer stages the runtime after makensis')
+ assert.match(
+ installer,
+ /File \/oname=VC_redist\.x64\.exe "\.\.\\\.\.\\\.build\\vc-redist\\VC_redist\.x64\.exe"/
+ )
+ assert.match(installer, /Delete "\$PLUGINSDIR\\VC_redist\.x64\.exe"/)
+ assertRuntimeExitHandling(installer, '$R4')
+})
diff --git a/services/installer/nvpair-setup.nsi b/services/installer/nvpair-setup.nsi
index 4e707423..8bb6c8c1 100644
--- a/services/installer/nvpair-setup.nsi
+++ b/services/installer/nvpair-setup.nsi
@@ -13,6 +13,7 @@
!include "MUI2.nsh"
!include "FileFunc.nsh"
+!include "LogicLib.nsh"
;---------------------------------------
; General
@@ -131,6 +132,38 @@ FunctionEnd
Sleep 500
!macroend
+; Install the centrally serviced Microsoft runtime required by the managed
+; llama.cpp binaries. Microsoft's x64 package also contains the ARM64 runtime.
+!macro InstallVcRuntime
+ DetailPrint "Installing Microsoft Visual C++ Runtime..."
+ ClearErrors
+ ExecWait '"$PLUGINSDIR\VC_redist.x64.exe" /install /quiet /norestart' $R4
+ ${if} ${Errors}
+ ClearErrors
+ Delete "$PLUGINSDIR\VC_redist.x64.exe"
+ MessageBox MB_OK|MB_ICONSTOP "${PRODUCT_DISPLAY_NAME} could not start the Microsoft Visual C++ Runtime installer.$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK
+ SetErrorLevel 3
+ Quit
+ ${endif}
+
+ Delete "$PLUGINSDIR\VC_redist.x64.exe"
+ ${if} $R4 == 0
+ DetailPrint "Microsoft Visual C++ Runtime installed."
+ ${elseIf} $R4 == 1638
+ DetailPrint "Microsoft Visual C++ Runtime is already installed."
+ ${elseIf} $R4 == 1641
+ DetailPrint "Microsoft Visual C++ Runtime installed; Windows restart initiated."
+ SetRebootFlag true
+ ${elseIf} $R4 == 3010
+ DetailPrint "Microsoft Visual C++ Runtime installed; Windows restart required."
+ SetRebootFlag true
+ ${else}
+ MessageBox MB_OK|MB_ICONSTOP "${PRODUCT_DISPLAY_NAME} could not install the Microsoft Visual C++ Runtime (exit code $R4).$\n$\nRestart Windows and run this installer again. If the problem continues, install the latest supported Visual C++ Redistributable from https://aka.ms/vc14/vc_redist.x64.exe, then retry." /SD IDOK
+ SetErrorLevel 3
+ Quit
+ ${endif}
+!macroend
+
;---------------------------------------
; Installer Section
;---------------------------------------
@@ -140,6 +173,10 @@ Section "Install"
; with a sharing violation and leave a half-installed mess.
!insertmacro CloseRunningInstance
+ SetOutPath "$PLUGINSDIR"
+ File /oname=VC_redist.x64.exe "..\..\.build\vc-redist\VC_redist.x64.exe"
+ !insertmacro InstallVcRuntime
+
; If .onInit found a previous install, remove it now — at this point the
; user has clicked Install, so the destructive step is safe.
StrCmp $PrevUninstaller "" skip_prev_uninstall
diff --git a/services/installer_build.bat b/services/installer_build.bat
index acb045f5..f6d2e498 100644
--- a/services/installer_build.bat
+++ b/services/installer_build.bat
@@ -31,7 +31,21 @@ if %ERRORLEVEL% neq 0 (
echo.
echo ========================================
-echo Step 2: Creating installer
+echo Step 2: Staging Visual C++ Runtime
+echo ========================================
+echo.
+
+node "%ROOT%..\scripts\stage-vc-redist.mjs"
+if errorlevel 1 (
+ echo.
+ echo Visual C++ Runtime staging failed -- aborting installer creation.
+ endlocal
+ exit /b 1
+)
+
+echo.
+echo ========================================
+echo Step 3: Creating installer
echo ========================================
echo.
diff --git a/services/nvpair-engine-manager/LAUNCH_TEXT.md b/services/nvpair-engine-manager/LAUNCH_TEXT.md
index 12a59b91..c1772df7 100644
--- a/services/nvpair-engine-manager/LAUNCH_TEXT.md
+++ b/services/nvpair-engine-manager/LAUNCH_TEXT.md
@@ -149,15 +149,17 @@ The existing engine-manager suite also exercises the JSON-RPC stdio path.
## Networking adapter coverage
-The bundled engines are Ollama and LM Studio. The shared parser contains no engine
-name checks. Their mandatory networking declarations and all platform variants
-are pinned by TestBundledNetworkingControls; adding an engine requires extending
-that inventory and reviewing its networking alternatives. Other settings have no catalog.
+The bundled engines are Ollama, LM Studio, and llama.cpp. The shared parser
+contains no engine name checks. Their mandatory networking declarations and all
+platform variants are pinned by TestBundledNetworkingControls; adding an engine
+requires extending that inventory and reviewing its networking alternatives.
+Other settings have no catalog.
| Engine | Port/bind controls | CORS controls | Vendor reference |
| --- | --- | --- | --- |
| Ollama | OLLAMA_HOST | OLLAMA_ORIGINS, including vendor quote stripping | [Environment source](https://github.com/ollama/ollama/blob/main/envconfig/config.go) |
| LM Studio | --port / -p; --bind / LMS_SERVER_HOST | --cors (no short alias) | [Server command source](https://github.com/lmstudio-ai/lms/blob/main/src/subcommands/server.ts) |
+| llama.cpp | --host; --port | --cors-origins | [Server reference](https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md) |
Remote CORS comparisons use normalized policy. The broker derives the internal
preserveCORS preview guard from the authenticated caller and repeats validation
diff --git a/services/nvpair-engine-manager/MANIFEST.md b/services/nvpair-engine-manager/MANIFEST.md
index a5f13515..c7282520 100644
--- a/services/nvpair-engine-manager/MANIFEST.md
+++ b/services/nvpair-engine-manager/MANIFEST.md
@@ -141,7 +141,7 @@ and recovery. Editing `args`/`start` directly remains trusted manifest authoring
|---|---|---|---|
| `fetch.url` | string | when `fetch` present | Download URL — **HTTPS** (plain `http` only from loopback). |
| `fetch.sha256` | string | no | Hex SHA-256. When set, the download is verified against it **before** `run` executes; when omitted, the fetch is HTTPS-only and runs with a loud "unpinned" warning (the same weaker guarantee as `script`). Pin it for any real release. |
-| `run` | string[] | no | Argv to execute after download (e.g. run the installer, extract the archive). Placeholders resolved; OS env refs expanded. Requires a `fetch` (the artifact it unpacks). |
+| `run` | string[] | when `fetch` or nonempty `artifacts` present | Nonempty argv to execute after download (e.g. run the installer, extract the archive). Placeholders resolved; OS env refs expanded. The child also receives exact paths in `NVPAIR_INSTALL_DIR`, `NVPAIR_INSTALL_DOWNLOAD`, and `NVPAIR_INSTALL_DOWNLOAD_` so commands that reparse argv can avoid shell quoting. Requires a `fetch` or `artifacts`. |
| `script` | string[] | no | **Escape hatch** for vendors that only ship a script installer. Runs **without** checksum verification (logged as unpinned) and replaces `fetch`+`run`. Prefer `fetch`+`run` whenever the vendor publishes a script or artifact: download it first, then execute the local file. **Make failures loud:** a piped bootstrap such as `curl … \| bash` can mask a failed fetch, while a separate fetch prevents the run and reports the error. |
| `mode` | string | no | `"user"` (default) or `"admin"`. The runner **refuses** `"admin"` (engine-manager is user-mode only); it is a deliberate, flagged exception, not a default. |
@@ -163,7 +163,11 @@ and recovery. Editing `args`/`start` directly remains trusted manifest authoring
A **probe** is `{ "http": "", "status": , "timeout_s": , "interval_s": }`
or `{ "tcp": "", ... }`. `status` defaults to `200`. Prefer
-loopback URLs/addresses.
+loopback URLs/addresses. An HTTP probe may also set
+`"json_match":{"field":"service.role","value":"router"}` to require a JSON
+string at that dotted object path. Missing, malformed, oversized, or
+wrong-typed response data fails the probe. `json_match` is invalid on TCP
+probes.
### Actions
@@ -171,8 +175,10 @@ Each action is a config-declared operation exposed over `engine:action`.
Exactly one of `http`, `cmd`, or `remove_path`:
- **`http`** — call the engine's loopback control API. The caller's
- `params` are sent as the JSON request body; `body_schema` is
- informational. Requires the engine to be running.
+ `params` are sent as the JSON request body by default. Set
+ `params_in: "query"` to require a JSON object of string values and URL-encode
+ them into the query string instead. `body_schema` is informational. Requires
+ the engine to be running.
- **`cmd`** — run a CLI command (e.g. `lms get`). The caller's `params`
become placeholders (e.g. `{model}`); stdout is returned (parsed as
JSON when it is valid JSON). Does **not** require the engine to be
@@ -302,23 +308,28 @@ validation at load:
| `{bin}` | Resolved binary path (process mode) | runtime args/env |
| `{cli}` | The platform's `runtime.cli` path | runtime start/stop, action `cmd` |
| `{download}` | Path of the verified download | `install.run` |
+| `{download_}` | Path of a verified member of `install.artifacts` | `install.run` |
| `{install_dir}` | Per-engine user-scoped install dir | `detect`, `install`, runtime |
A `cmd` action additionally templates the action's own `params` as
placeholders (e.g. `{model}`), resolved at call time. HTTP actions send
-`params` as the JSON request **body** — they are not substituted into
-`http.path`, which templates only `{port}`.
+`params` as the JSON request **body** by default; `params_in: "query"` sends
+their string fields as URL-encoded query parameters with no body. They are not
+substituted into `http.path`, which templates only `{port}`.
## Validation
A manifest is rejected at load (with a specific message) when: a required
field is missing, `manifest_version` is unsupported, a platform key isn't
`"/"`, `runtime.bin` is empty in process mode (or
-`runtime.start` is empty in command mode), `install.run` has no `fetch`,
-`install.script` is combined with `fetch`/`run`, `install.mode` or
+`runtime.start` is empty in command mode), `install.run` has neither `fetch`
+nor nonempty `artifacts`, `fetch` or nonempty `artifacts` is present without
+nonempty `install.run`, `install.script` is combined with
+`fetch`/`artifacts`/`run`, `install.mode` or
`runtime.mode` is invalid, an action sets none or more than one of
`http`/`cmd`/`remove_path`, a `remove_path` action omits `root` or
-`path`, a `result` is set without both `array` and `field` (or a
+`path`, `http.params_in` is not `body` or `query`, a `result` is set without
+both `array` and `field` (or a
`result.match` without both `match.field` and a non-empty `match.in`), or
a non-action templated string uses an unknown placeholder.
@@ -430,7 +441,7 @@ GPU selection, and auth.
| Engine | Fit | Headless launch | Config surface | Control |
|---|---|---|---|---|
| **Ollama** | strong (env-first) | `ollama serve` (foreground) | env: `OLLAMA_HOST`, `OLLAMA_MODELS`, `OLLAMA_KEEP_ALIVE`, `OLLAMA_NUM_PARALLEL`, `OLLAMA_MAX_LOADED_MODELS`, `OLLAMA_MAX_QUEUE`, `OLLAMA_CONTEXT_LENGTH`, `OLLAMA_FLASH_ATTENTION` | HTTP `/api/tags`, `/api/pull`; CLI `ollama pull/ls/ps/stop` |
-| **llama.cpp** | strong (env+flags) — **reference design** | `llama-server --host 127.0.0.1 --port {port}` (foreground) | flags + `LLAMA_ARG_*` (host/port, ctx-size, n-parallel, cont-batching, flash-attn, device, n-gpu-layers, tensor-split, main-gpu, api-key, models-dir/max) | HTTP `/v1/models`, `/models/load`, `/models/unload`, `/health`, `/slots` |
+| **llama.cpp** | strong (env+flags) — **reference design** | `llama-server --host 127.0.0.1 --port {port}` (foreground) | flags + `LLAMA_ARG_*` (host/port, ctx-size, n-parallel, cont-batching, flash-attn, device, n-gpu-layers, tensor-split, main-gpu, api-key, models-dir/max) | HTTP `/v1/models`, `/models` download/delete, `/models/load`, `/models/unload`, `/health`, `/slots` |
| **LM Studio** | command / daemon | `lms daemon up` → `lms server start --port {port}` | small env (`LMS_SERVER_HOST`, `LM_API_TOKEN`); most config is flags/API/settings (`lms load --context-length/--gpu/--ttl`) | HTTP `/api/v1/models[/download\|load\|unload]`; CLI `lms ls/get/load/unload/ps` |
| **vLLM** | flags-first — **Linux/WSL only** | `vllm serve --host 127.0.0.1 --port {port}` | flags (host/port/api-key); `HF_HOME` for cache. `VLLM_PORT`/`VLLM_HOST_IP` are **not** the API bind | OpenAI `/v1/models`; one model per process (unload = restart) |
| **Jan** | hybrid (on llama.cpp router) | `jan serve --port {port}` (CLI) | forwards `LLAMA_ARG_*`; perf settings are router-preset-driven | `jan serve` auto-downloads HF repos |
diff --git a/services/nvpair-engine-manager/README.md b/services/nvpair-engine-manager/README.md
index b401c883..8511e2a3 100644
--- a/services/nvpair-engine-manager/README.md
+++ b/services/nvpair-engine-manager/README.md
@@ -5,15 +5,38 @@ SPDX-License-Identifier: Apache-2.0
# nvpair-engine-manager
-A config-driven control plane for local inference engines (Ollama today;
-Intel/others via a dropped-in manifest). It manages everything about an
-engine **except serving inference**: detect, user-mode install,
+A config-driven control plane for local inference engines. The bundled
+manifests support Ollama, LM Studio, and llama.cpp. It manages everything about
+an engine **except serving inference**: detect, user-mode install,
start/stop/restart, health, and config-declared actions. Adding an engine
is a JSON manifest, not code.
The bundled manifests under `manifests/` are the working reference for manifest
authoring.
+### llama.cpp backend checkpoint
+
+The `llamacpp` manifest runs `llama-server` in router mode on loopback port
+`8081`. It can list, download, load, unload, and delete exact model ids such as
+`owner/repository:Q4_K_M`. Downloads use `/models/sse` for progress; deletion
+uses the router's native `DELETE /models` cache operation and does not restart
+the router. `LLAMA_CACHE` points to a managed sibling directory so models
+survive engine uninstall and reinstall.
+Readiness requires `/props` to report `role:"router"`, so model-selection
+arguments that switch `llama-server` to single-model mode and incompatible
+listeners already occupying the port are rejected rather than adopted.
+After five minutes without inference work, a loaded model enters llama.cpp sleep
+mode and releases its model and KV-cache memory; the next request wakes it. The
+router child remains alive and can retain a residual backend GPU context.
+
+Windows and Linux installs download checksum-pinned server and CUDA-runtime
+archive pairs (CUDA 12.x for x64 and CUDA 13.4 for arm64); macOS uses the
+standard Metal-capable archive. These on-demand downloads are roughly
+0.6–0.8 GiB and do not enlarge the PAIR installer. llama.cpp keeps its `auto`
+GPU-layer policy: supported NVIDIA/Metal devices can accelerate, while the
+dynamic CPU backend remains the fallback. Hardware acceptance, not `/health`
+alone, is required to claim GPU activation.
+
## Communication
Bidirectional newline-delimited JSON-RPC 2.0 — the same conventions as
@@ -71,6 +94,28 @@ UI converges even if its synchronous call already timed out),
pipeline — `errors:report` / `errors:clear` (consumed by `nvpair-errors`
via the broker; see below).
+Streaming HTTP pulls have a 30-minute **inactivity** watchdog. Each increase in
+an Ollama layer's or llama.cpp file's completed byte count refreshes that
+watchdog, so an active download may run longer than 30 minutes; duplicate
+progress frames and heartbeats do not extend a stalled pull. CLI-driven pulls
+without structured byte progress retain the fixed 30-minute action timeout.
+
+For llama.cpp, an accepted pull that ends before a matching `download_finished`
+or `download_failed` event stops the active download. This includes caller
+cancellation, remote caller disconnect, inactivity timeout, and premature SSE
+termination. Cleanup checks `GET /models` and sends `POST /models/unload` only
+when the exact model is still `downloading`; cached files are retained, and
+models that have already completed are left alone.
+
+The initial `POST /models` handshake has a separate 30-second total timeout and
+continues through caller cancellation so its acceptance can still be read.
+The pull then waits for cleanup, which has a separate five-second budget for
+the inventory check and stop request together. A failed cleanup reports that
+the download could not be confirmed stopped alongside the original pull error.
+If the start acknowledgement is lost or unreadable, acceptance and cancellation
+are reported as unconfirmed without unloading an unowned download. The monitored
+model must match `params.model`; mismatches are rejected before subscribing.
+
The `engine:remote-*` methods are the client half of remote engine
management: engine-manager resolves the target `node` in an `ec` peer
directory (fed by its own `discovery:subscribe{services:[ec]}` to the broker
@@ -117,7 +162,7 @@ full authoritative snapshots to pinned peers.
```
NotInstalled --engine:install--> (HTTPS download + verify-if-pinned + user-mode run) --> Stopped
Stopped --engine:start----> (adopt if already serving the port, else spawn) --> Running --health--> Running
-Running --engine:stop-----> (stop signal, wait for exit; no timeout) --> Stopped
+Running --engine:stop-----> (stop signal, bounded grace, force if needed) --> Stopped
```
Detect uses the manifest's `detect` paths. Install is one-shot and
@@ -128,16 +173,18 @@ unexpected exit is reported. The bundled Ollama manifest allows up to ten
minutes for startup because GPU discovery can exceed the previous 30-second
allowance on supported Windows systems. The deadline remains finite: if Ollama
never serves its readiness endpoint, engine-manager stops the owned process and
-reports the failed start. Stop sends one stop signal and waits for the engine
-to exit, with no timeout: SIGTERM to the process group on Unix (graceful, no
-SIGKILL escalation), and `taskkill /T /F` on Windows — where the windowless
-engines we spawn can't receive a graceful (non-`/F`) close, so a forced
-terminate is the only signal that actually stops them.
+reports the failed start. On Unix, stopping an owned process sends SIGTERM to
+its process group, waits the manifest's `stop.grace_s` (five seconds by
+default), then escalates to SIGKILL; `signal:"kill"` skips the grace. Failed
+startup cleanup uses the same policy. On Windows, the windowless engines we
+spawn cannot receive a graceful (non-`/F`) close, so stopping uses immediate
+`taskkill /T /F`.
### Adoption — start may attach to an engine it didn't launch
Before spawning, `engine:start` **probes the chosen port's readiness
-endpoint**. If something already answers there — the engine's own desktop
+endpoint**, including any manifest-declared JSON identity. If a compatible
+service answers there — the engine's own desktop
app (e.g. the Ollama tray app on `11434`), or an instance left running from a
previous session — the service **adopts** that instance: it marks the engine
`running` without launching its own, rather than spawning a duplicate that
diff --git a/services/nvpair-engine-manager/actions.go b/services/nvpair-engine-manager/actions.go
index 372ee104..8fcb4b08 100644
--- a/services/nvpair-engine-manager/actions.go
+++ b/services/nvpair-engine-manager/actions.go
@@ -11,6 +11,7 @@ import (
"io"
"log/slog"
"net/http"
+ "net/url"
"os"
"os/exec"
"path/filepath"
@@ -83,7 +84,7 @@ func (e *Executor) restartAfterAction(ctx context.Context, st *engineState, engi
// dispatchAction invokes the action itself: a guarded filesystem removal, a CLI
// command, or an HTTP call against the engine's loopback control API with the
-// caller's params as the body.
+// caller's params in the manifest-declared location.
func (e *Executor) dispatchAction(ctx context.Context, st *engineState, engine, action string, act Action, params json.RawMessage) (json.RawMessage, error) {
ctx, cancel := context.WithTimeout(ctx, e.actionTimeout)
defer cancel()
@@ -111,13 +112,30 @@ func (e *Executor) dispatchAction(ctx context.Context, st *engineState, engine,
if err != nil {
return nil, err
}
- url := fmt.Sprintf("http://127.0.0.1:%d%s", port, path)
+ requestURL := fmt.Sprintf("http://127.0.0.1:%d%s", port, path)
var body io.Reader
- if len(params) > 0 && string(params) != "null" {
+ if act.HTTP.ParamsIn == actionHTTPParamsQuery {
+ parsedURL, err := url.Parse(requestURL)
+ if err != nil {
+ return nil, fmt.Errorf("action %q: invalid HTTP URL: %w", action, err)
+ }
+ queryParams, err := actionQueryParams(params)
+ if err != nil {
+ return nil, fmt.Errorf("action %q: %w", action, err)
+ }
+ query := parsedURL.Query()
+ for name, values := range queryParams {
+ for _, value := range values {
+ query.Add(name, value)
+ }
+ }
+ parsedURL.RawQuery = query.Encode()
+ requestURL = parsedURL.String()
+ } else if len(params) > 0 && string(params) != "null" {
body = bytes.NewReader(params)
}
- req, err := http.NewRequestWithContext(ctx, strings.ToUpper(act.HTTP.Method), url, body)
+ req, err := http.NewRequestWithContext(ctx, strings.ToUpper(act.HTTP.Method), requestURL, body)
if err != nil {
return nil, err
}
@@ -148,6 +166,21 @@ func (e *Executor) dispatchAction(ctx context.Context, st *engineState, engine,
return wrapped, nil
}
+func actionQueryParams(params json.RawMessage) (url.Values, error) {
+ query := make(url.Values)
+ if len(params) == 0 {
+ return query, nil
+ }
+ var values map[string]string
+ if err := json.Unmarshal(params, &values); err != nil {
+ return nil, fmt.Errorf("http query params must be a JSON object with string values: %w", err)
+ }
+ for name, value := range values {
+ query.Add(name, value)
+ }
+ return query, nil
+}
+
// runRemovePathAction resolves templated path/root placeholders and deletes
// the target when it stays under the declared root.
func (e *Executor) runRemovePathAction(ctx context.Context, st *engineState, act Action, params json.RawMessage) (json.RawMessage, error) {
diff --git a/services/nvpair-engine-manager/e2e_test.go b/services/nvpair-engine-manager/e2e_test.go
index e02187a0..a86ac5a5 100644
--- a/services/nvpair-engine-manager/e2e_test.go
+++ b/services/nvpair-engine-manager/e2e_test.go
@@ -241,9 +241,9 @@ type e2eManager struct {
stopped bool
}
-func startE2EManager(t *testing.T, cfg, home string) *e2eManager {
+func startE2EManager(t *testing.T, cfg, home string, args ...string) *e2eManager {
t.Helper()
- cmd := exec.Command(managerBin)
+ cmd := exec.Command(managerBin, args...)
cmd.Env = overrideEnv(map[string]string{"APPDATA": cfg, "LOCALAPPDATA": cfg, "XDG_CONFIG_HOME": cfg, "HOME": home})
stdin, err := cmd.StdinPipe()
if err != nil {
diff --git a/services/nvpair-engine-manager/executor.go b/services/nvpair-engine-manager/executor.go
index de6bcfdd..2f80b16d 100644
--- a/services/nvpair-engine-manager/executor.go
+++ b/services/nvpair-engine-manager/executor.go
@@ -19,7 +19,12 @@ import (
"time"
)
-var winEnvRe = regexp.MustCompile(`%([^%]+)%`)
+var (
+ winEnvRe = regexp.MustCompile(`%([^%]+)%`)
+ // Match named variables only so shell parameters such as $1 and $@ reach
+ // manifest commands unchanged.
+ unixEnvRe = regexp.MustCompile(`\$(?:\{[A-Za-z_][A-Za-z0-9_]*\}|[A-Za-z_][A-Za-z0-9_]*)`)
+)
const (
engineResponseHeaderTimeout = 30 * time.Second
@@ -93,9 +98,18 @@ type Executor struct {
// detectTimeout bounds the post-install/uninstall detect poll
// (installers finish their file work asynchronously). Overridable.
detectTimeout time.Duration
- // actionTimeout bounds a single engine:action call (HTTP or CLI) so a
- // hung engine can't park the goroutine or starve the caller forever.
+ // actionTimeout bounds ordinary engine actions and CLI-driven model pulls so
+ // a hung engine can't park the goroutine or starve the caller forever.
actionTimeout time.Duration
+ // pullProgressTimeout bounds how long a streaming HTTP model pull may go
+ // without advancing a layer/file byte count. Advancing progress refreshes
+ // the deadline, allowing large active downloads to exceed actionTimeout.
+ pullProgressTimeout time.Duration
+ // pullStartTimeout lets an asynchronous pull finish its start handshake even
+ // after caller cancellation, so an accepted download can still be stopped.
+ pullStartTimeout time.Duration
+ // pullCleanupTimeout bounds the inventory check and download stop together.
+ pullCleanupTimeout time.Duration
// loadedPollInterval is the cadence of the loaded-model watcher
// (loadedwatch.go), which polls each running engine's resident set and emits
// engine:models-changed on change. 0 disables it. Overridable via
@@ -115,19 +129,22 @@ type Executor struct {
func NewExecutor(reg *Registry, reporter *Reporter, emit func(string, any), baseDir string) *Executor {
return &Executor{
- reg: reg,
- reporter: reporter,
- emit: emit,
- client: newEngineHTTPClient(engineResponseHeaderTimeout),
- ollamaLoadClient: newEngineHTTPClient(ollamaLoadResponseHeaderTimeout),
- progress: newProgressHub(),
- baseDir: baseDir,
- desired: newDesiredStateStore(baseDir),
- detectTimeout: 30 * time.Second,
- actionTimeout: 30 * time.Minute,
- loadedPollInterval: defaultLoadedPollSeconds * time.Second,
- loadedPoke: make(chan struct{}, 1),
- engines: make(map[string]*engineState),
+ reg: reg,
+ reporter: reporter,
+ emit: emit,
+ client: newEngineHTTPClient(engineResponseHeaderTimeout),
+ ollamaLoadClient: newEngineHTTPClient(ollamaLoadResponseHeaderTimeout),
+ progress: newProgressHub(),
+ baseDir: baseDir,
+ desired: newDesiredStateStore(baseDir),
+ detectTimeout: 30 * time.Second,
+ actionTimeout: 30 * time.Minute,
+ pullProgressTimeout: 30 * time.Minute,
+ pullStartTimeout: 30 * time.Second,
+ pullCleanupTimeout: 5 * time.Second,
+ loadedPollInterval: defaultLoadedPollSeconds * time.Second,
+ loadedPoke: make(chan struct{}, 1),
+ engines: make(map[string]*engineState),
}
}
@@ -151,7 +168,8 @@ func (e *Executor) reservedPortError(port int) error {
// Ollama's cold-load action gets a separate 10m client. Both set NO total
// http.Client.Timeout: a multi-GB engine download can legitimately run
// for many minutes and every call site already bounds total time with a
-// context deadline (download 30m, action actionTimeout, probe 3s). What
+// context deadline or inactivity watchdog (install download 30m, streaming
+// pull pullProgressTimeout, action actionTimeout, probe 3s). What
// it adds over the zero-value client is (a) a bounded response-header
// wait so a peer that accepts the connection but never replies can't park
// a goroutine even inside a long context, and (b) a redirect policy that
@@ -255,7 +273,13 @@ func expandPathForOS(s, goos string) string {
return os.Getenv(tok[1 : len(tok)-1])
})
} else {
- s = os.ExpandEnv(s)
+ s = unixEnvRe.ReplaceAllStringFunc(s, func(tok string) string {
+ name := tok[1:]
+ if name[0] == '{' {
+ name = name[1 : len(name)-1]
+ }
+ return os.Getenv(name)
+ })
}
switch {
case s == "~":
diff --git a/services/nvpair-engine-manager/executor_test.go b/services/nvpair-engine-manager/executor_test.go
index 40ebe249..f15fd99a 100644
--- a/services/nvpair-engine-manager/executor_test.go
+++ b/services/nvpair-engine-manager/executor_test.go
@@ -8,6 +8,7 @@ import (
"crypto/sha256"
"encoding/hex"
"encoding/json"
+ "io"
"net"
"net/http"
"net/http/httptest"
@@ -134,6 +135,114 @@ func TestOnlyOllamaRunModelUsesSlowResponseHeaderBudget(t *testing.T) {
}
}
+func TestHTTPActionSendsStringParamsInQuery(t *testing.T) {
+ type observedRequest struct {
+ method string
+ rawQuery string
+ model string
+ contentType string
+ body []byte
+ readErr error
+ }
+
+ observed := make(chan observedRequest, 1)
+ srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ body, err := io.ReadAll(r.Body)
+ observed <- observedRequest{
+ method: r.Method,
+ rawQuery: r.URL.RawQuery,
+ model: r.URL.Query().Get("model"),
+ contentType: r.Header.Get("Content-Type"),
+ body: body,
+ readErr: err,
+ }
+ w.Header().Set("Content-Type", "application/json")
+ _, _ = w.Write([]byte(`{"success":true}`))
+ }))
+ defer srv.Close()
+
+ serverURL, err := url.Parse(srv.URL)
+ if err != nil {
+ t.Fatalf("parse test server URL: %v", err)
+ }
+ port, err := strconv.Atoi(serverURL.Port())
+ if err != nil {
+ t.Fatalf("parse test server port: %v", err)
+ }
+
+ m := testEngineManifest(fakeEngineBin)
+ m.Engine = "llamacpp"
+ platform := m.Platforms[runtime.GOOS+"/"+runtime.GOARCH]
+ platform.Runtime.Port = port
+ m.Platforms[runtime.GOOS+"/"+runtime.GOARCH] = platform
+ m.Actions = map[string]Action{
+ "delete_model": {
+ HTTP: &ActionHTTP{
+ Method: http.MethodDelete,
+ Path: "/models",
+ ParamsIn: actionHTTPParamsQuery,
+ },
+ },
+ }
+
+ ex := newTestExecutor(t, m)
+ st, err := ex.state("llamacpp")
+ if err != nil {
+ t.Fatalf("state: %v", err)
+ }
+ st.running = true
+
+ const model = "owner/repository:Q4_K_M"
+ res, err := ex.Action(
+ context.Background(),
+ "llamacpp",
+ "delete_model",
+ json.RawMessage(`{"model":"owner/repository:Q4_K_M"}`),
+ )
+ if err != nil {
+ t.Fatalf("delete_model: %v", err)
+ }
+ var result struct {
+ Success bool `json:"success"`
+ }
+ if err := json.Unmarshal(res, &result); err != nil {
+ t.Fatalf("decode delete response: %v", err)
+ }
+ if !result.Success {
+ t.Fatalf("delete response = %s, want success", res)
+ }
+
+ got := <-observed
+ if got.readErr != nil {
+ t.Fatalf("read request body: %v", got.readErr)
+ }
+ if got.method != http.MethodDelete {
+ t.Errorf("method = %q, want %q", got.method, http.MethodDelete)
+ }
+ if got.model != model {
+ t.Errorf("model query = %q, want %q", got.model, model)
+ }
+ if got.rawQuery != "model=owner%2Frepository%3AQ4_K_M" {
+ t.Errorf("raw query = %q, want URL-encoded model id", got.rawQuery)
+ }
+ if len(got.body) != 0 {
+ t.Errorf("body = %q, want empty", got.body)
+ }
+ if got.contentType != "" {
+ t.Errorf("Content-Type = %q, want empty without a body", got.contentType)
+ }
+}
+
+func TestHTTPActionRejectsNonStringQueryParams(t *testing.T) {
+ _, err := actionQueryParams(json.RawMessage(`{"model":42}`))
+ if err == nil {
+ t.Fatal("non-string query param accepted")
+ }
+ if !strings.Contains(err.Error(), "string values") {
+ t.Fatalf("error = %q, want string-values requirement", err)
+ }
+}
+
func TestEngineLifecycle(t *testing.T) {
ex := newTestExecutor(t, testEngineManifest(fakeEngineBin))
ctx := context.Background()
diff --git a/services/nvpair-engine-manager/health_connections_test.go b/services/nvpair-engine-manager/health_connections_test.go
index 44e13a3a..d613f5cb 100644
--- a/services/nvpair-engine-manager/health_connections_test.go
+++ b/services/nvpair-engine-manager/health_connections_test.go
@@ -51,6 +51,32 @@ func TestProbeHTTPReusesConnections(t *testing.T) {
test("chunked", true)
}
+func TestProbeHTTPMatchesJSONIdentity(t *testing.T) {
+ test := func(name, body string, want bool) {
+ t.Helper()
+ t.Run(name, func(t *testing.T) {
+ client, _ := testclient.New(t, http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Type", "application/json")
+ _, _ = io.WriteString(w, body)
+ }))
+ ex := &Executor{client: client}
+ probe := &Probe{
+ HTTP: "http://127.0.0.1:{port}/props",
+ JSONMatch: &ProbeJSONMatch{Field: "service.role", Value: "router"},
+ }
+ if got := ex.probe(context.Background(), probe, 1); got != want {
+ t.Fatalf("probe = %v, want %v", got, want)
+ }
+ })
+ }
+
+ test("matching nested string", `{"service":{"role":"router"}}`, true)
+ test("different string", `{"service":{"role":"worker"}}`, false)
+ test("missing field", `{"service":{}}`, false)
+ test("wrong field type", `{"service":{"role":true}}`, false)
+ test("malformed JSON", `{"service":`, false)
+}
+
func TestProbeHTTPBoundsBodyDrain(t *testing.T) {
body := &healthProbeBody{reader: strings.NewReader(strings.Repeat("x", 4<<20))}
ex := &Executor{client: &http.Client{Transport: healthProbeTransport(func(*http.Request) (*http.Response, error) {
@@ -65,6 +91,23 @@ func TestProbeHTTPBoundsBodyDrain(t *testing.T) {
}
}
+func TestProbeHTTPRejectsOversizedJSONIdentityBody(t *testing.T) {
+ body := &healthProbeBody{reader: strings.NewReader(strings.Repeat("x", 4<<20))}
+ ex := &Executor{client: &http.Client{Transport: healthProbeTransport(func(*http.Request) (*http.Response, error) {
+ return &http.Response{StatusCode: http.StatusOK, Body: body, Header: make(http.Header)}, nil
+ })}}
+ probe := &Probe{
+ HTTP: "http://127.0.0.1:{port}/",
+ JSONMatch: &ProbeJSONMatch{Field: "role", Value: "router"},
+ }
+ if ex.probe(context.Background(), probe, 1) {
+ t.Fatal("oversized JSON identity body passed the probe")
+ }
+ if body.read == 0 || body.read > 2*maxProbeJSONBytes+1 || !body.closed {
+ t.Fatalf("body read = %d, closed = %v; want bounded read and close", body.read, body.closed)
+ }
+}
+
func TestProbeHTTPBodyDrainHonorsDeadline(t *testing.T) {
var body *healthProbeBody
ex := &Executor{client: &http.Client{Transport: healthProbeTransport(func(req *http.Request) (*http.Response, error) {
diff --git a/services/nvpair-engine-manager/install.go b/services/nvpair-engine-manager/install.go
index 268bd0b9..e3643d6c 100644
--- a/services/nvpair-engine-manager/install.go
+++ b/services/nvpair-engine-manager/install.go
@@ -96,6 +96,18 @@ func (e *Executor) Install(ctx context.Context, engine string) error {
vars["download"] = dp
e.emitInstallProgress(engine, "verified", 50)
}
+ if len(inst.Artifacts) > 0 {
+ paths, artifactVars, err := e.downloadInstallArtifacts(ctx, engine, inst.Artifacts)
+ defer removeDownloadedFiles(paths)
+ if err != nil {
+ e.reportInstallFailed(engine, err)
+ return err
+ }
+ for name, downloadPath := range artifactVars {
+ vars[name] = downloadPath
+ }
+ e.emitInstallProgress(engine, "verified", 50)
+ }
if len(inst.Run) > 0 {
e.emitInstallProgress(engine, "installing", 75)
args, err := resolveArgs(inst.Run, vars)
@@ -106,7 +118,7 @@ func (e *Executor) Install(ctx context.Context, engine string) error {
for i := range args {
args[i] = expandPath(args[i])
}
- if err := e.runCommand(ctx, args); err != nil {
+ if err := e.runCommandWithEnv(ctx, args, installCommandEnv(vars, inst.Artifacts)); err != nil {
werr := fmt.Errorf("install command failed: %w", err)
e.reportInstallFailed(engine, werr)
return werr
@@ -234,6 +246,9 @@ func validateDownloadURL(raw string) error {
if err != nil {
return fmt.Errorf("invalid download url %q: %w", raw, err)
}
+ if u.Hostname() == "" {
+ return fmt.Errorf("download url %q must include a host", raw)
+ }
switch u.Scheme {
case "https":
return nil
@@ -248,6 +263,64 @@ func validateDownloadURL(raw string) error {
}
func (e *Executor) download(ctx context.Context, engine string, f *Fetch) (string, error) {
+ return e.downloadWithProgress(ctx, engine, f, func(percent int) {
+ e.emitInstallProgress(engine, "downloading", percent)
+ })
+}
+
+func (e *Executor) downloadInstallArtifacts(
+ ctx context.Context,
+ engine string,
+ artifacts []InstallArtifact,
+) ([]string, map[string]string, error) {
+ paths := make([]string, 0, len(artifacts))
+ vars := make(map[string]string, len(artifacts))
+ e.emitInstallProgress(engine, "downloading", 0)
+ for index, artifact := range artifacts {
+ fetch := &Fetch{URL: artifact.URL, SHA256: artifact.SHA256}
+ downloadPath, err := e.downloadWithProgress(ctx, engine, fetch, func(percent int) {
+ e.emitInstallProgress(engine, "downloading", aggregateArtifactProgress(index, len(artifacts), percent))
+ })
+ if err != nil {
+ return paths, vars, fmt.Errorf("download artifact %q: %w", artifact.Name, err)
+ }
+ paths = append(paths, downloadPath)
+ vars["download_"+artifact.Name] = downloadPath
+ e.emitInstallProgress(engine, "downloading", aggregateArtifactProgress(index, len(artifacts), 100))
+ }
+ return paths, vars, nil
+}
+
+func aggregateArtifactProgress(index, count, percent int) int {
+ percent = max(0, min(percent, 100))
+ start := index * 50 / count
+ end := (index + 1) * 50 / count
+ return start + (end-start)*percent/100
+}
+
+func removeDownloadedFiles(paths []string) {
+ for _, downloadPath := range paths {
+ _ = os.Remove(downloadPath)
+ }
+}
+
+func installCommandEnv(vars map[string]string, artifacts []InstallArtifact) map[string]string {
+ env := map[string]string{"NVPAIR_INSTALL_DIR": vars["install_dir"]}
+ if downloadPath := vars["download"]; downloadPath != "" {
+ env["NVPAIR_INSTALL_DOWNLOAD"] = downloadPath
+ }
+ for _, artifact := range artifacts {
+ env["NVPAIR_INSTALL_DOWNLOAD_"+strings.ToUpper(artifact.Name)] = vars["download_"+artifact.Name]
+ }
+ return env
+}
+
+func (e *Executor) downloadWithProgress(
+ ctx context.Context,
+ engine string,
+ f *Fetch,
+ onProgress func(int),
+) (string, error) {
if err := validateDownloadURL(f.URL); err != nil {
return "", err
}
@@ -273,9 +346,7 @@ func (e *Executor) download(ctx context.Context, engine string, f *Fetch) (strin
if err != nil {
return "", err
}
- pw := &progressWriter{total: resp.ContentLength, onPct: func(p int) {
- e.emitInstallProgress(engine, "downloading", p)
- }}
+ pw := &progressWriter{total: resp.ContentLength, onPct: onProgress}
h := sha256.New()
// Read one byte past the cap so we can detect (and reject) overflow.
n, err := io.Copy(io.MultiWriter(tmp, h), io.TeeReader(io.LimitReader(resp.Body, maxDownloadBytes+1), pw))
@@ -310,10 +381,20 @@ func (e *Executor) download(ctx context.Context, engine string, f *Fetch) (strin
// step), hiding the console window on Windows; on failure it returns the
// combined output for diagnostics.
func (e *Executor) runCommand(ctx context.Context, argv []string) error {
+ return e.runCommandWithEnv(ctx, argv, nil)
+}
+
+func (e *Executor) runCommandWithEnv(ctx context.Context, argv []string, environment map[string]string) error {
if len(argv) == 0 {
return nil
}
cmd := exec.CommandContext(ctx, argv[0], argv[1:]...)
+ if len(environment) > 0 {
+ cmd.Env = os.Environ()
+ for key, value := range environment {
+ cmd.Env = append(cmd.Env, key+"="+value)
+ }
+ }
configureSysProcAttr(cmd) // hide the console window on Windows
out, err := cmd.CombinedOutput()
if err != nil {
diff --git a/services/nvpair-engine-manager/install_artifacts_test.go b/services/nvpair-engine-manager/install_artifacts_test.go
new file mode 100644
index 00000000..6f362906
--- /dev/null
+++ b/services/nvpair-engine-manager/install_artifacts_test.go
@@ -0,0 +1,209 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package main
+
+import (
+ "context"
+ "crypto/sha256"
+ "encoding/hex"
+ "encoding/json"
+ "errors"
+ "net/http"
+ "net/http/httptest"
+ "os"
+ "path/filepath"
+ "strings"
+ "sync"
+ "testing"
+ "time"
+)
+
+type installProgressRecorder struct {
+ mu sync.Mutex
+ events []any
+}
+
+func (r *installProgressRecorder) emit(method string, params any) {
+ if method != "engine:install-progress" {
+ return
+ }
+ r.mu.Lock()
+ r.events = append(r.events, params)
+ r.mu.Unlock()
+}
+
+func (r *installProgressRecorder) assertSingleTerminal(t *testing.T, want string) {
+ t.Helper()
+ r.mu.Lock()
+ events := append([]any(nil), r.events...)
+ r.mu.Unlock()
+
+ stages := make([]string, 0, len(events))
+ terminal := make([]string, 0, 1)
+ for index, params := range events {
+ event, ok := params.(map[string]any)
+ if !ok {
+ t.Fatalf("install progress event %d params type = %T, want map[string]any", index, params)
+ }
+ stage, ok := event["stage"].(string)
+ if !ok {
+ t.Fatalf("install progress event %d stage = %v, want string", index, event["stage"])
+ }
+ stages = append(stages, stage)
+ switch stage {
+ case "done", "already-installed", "failed":
+ terminal = append(terminal, stage)
+ }
+ }
+ if len(terminal) != 1 || terminal[0] != want {
+ t.Fatalf("terminal install progress stages = %v, want [%s]; all stages = %v", terminal, want, stages)
+ }
+}
+
+func TestInstallDownloadsAllNamedArtifactsBeforeRunning(t *testing.T) {
+ payload := []byte("artifact")
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ _, _ = w.Write(payload)
+ }))
+ defer server.Close()
+
+ marker := filepath.Join(t.TempDir(), "installed.json")
+ artifacts := []InstallArtifact{
+ pinnedArtifact("server", server.URL+"/server.zip", payload),
+ pinnedArtifact("cudart", server.URL+"/cudart.zip", payload),
+ }
+ manifest := artifactInstallManifest(t, "artifact-success", marker, artifacts)
+ executor := newTestExecutor(t, manifest)
+ progress := &installProgressRecorder{}
+ executor.emit = progress.emit
+
+ if err := executor.Install(context.Background(), manifest.Engine); err != nil {
+ t.Fatalf("install named artifacts: %v", err)
+ }
+ progress.assertSingleTerminal(t, "done")
+ data, err := os.ReadFile(marker)
+ if err != nil {
+ t.Fatalf("read captured install arguments: %v", err)
+ }
+ var downloads []string
+ if err := json.Unmarshal(data, &downloads); err != nil {
+ t.Fatalf("decode captured install arguments: %v", err)
+ }
+ if len(downloads) != 2 || downloads[0] == downloads[1] {
+ t.Fatalf("resolved artifact paths = %v", downloads)
+ }
+ assertNoArtifactTemps(t, manifest.Engine)
+}
+
+func TestInstallRejectsBadSecondArtifactBeforeCommand(t *testing.T) {
+ payload := []byte("artifact")
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ _, _ = w.Write(payload)
+ }))
+ defer server.Close()
+
+ marker := filepath.Join(t.TempDir(), "must-not-exist")
+ artifacts := []InstallArtifact{
+ pinnedArtifact("server", server.URL+"/server.zip", payload),
+ {Name: "cudart", URL: server.URL + "/cudart.zip", SHA256: strings.Repeat("0", 64)},
+ }
+ manifest := artifactInstallManifest(t, "artifact-bad-checksum", marker, artifacts)
+ executor := newTestExecutor(t, manifest)
+ progress := &installProgressRecorder{}
+ executor.emit = progress.emit
+
+ err := executor.Install(context.Background(), manifest.Engine)
+ if err == nil || !strings.Contains(err.Error(), "checksum mismatch") {
+ t.Fatalf("install error = %v, want checksum mismatch", err)
+ }
+ progress.assertSingleTerminal(t, "failed")
+ if fileExists(marker) {
+ t.Fatal("install command ran after an artifact checksum failed")
+ }
+ assertNoArtifactTemps(t, manifest.Engine)
+}
+
+func TestInstallCancellationRemovesDownloadedArtifacts(t *testing.T) {
+ firstPayload := []byte("server")
+ secondStarted := make(chan struct{}, 1)
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ if r.URL.Path == "/server.zip" {
+ _, _ = w.Write(firstPayload)
+ return
+ }
+ _, _ = w.Write([]byte("partial"))
+ if flusher, ok := w.(http.Flusher); ok {
+ flusher.Flush()
+ }
+ secondStarted <- struct{}{}
+ <-r.Context().Done()
+ }))
+ defer server.Close()
+
+ marker := filepath.Join(t.TempDir(), "must-not-exist")
+ artifacts := []InstallArtifact{
+ pinnedArtifact("server", server.URL+"/server.zip", firstPayload),
+ {Name: "cudart", URL: server.URL + "/cudart.zip", SHA256: strings.Repeat("c", 64)},
+ }
+ manifest := artifactInstallManifest(t, "artifact-cancel", marker, artifacts)
+ executor := newTestExecutor(t, manifest)
+ progress := &installProgressRecorder{}
+ executor.emit = progress.emit
+ ctx, cancel := context.WithCancel(context.Background())
+ done := make(chan error, 1)
+ go func() {
+ done <- executor.Install(ctx, manifest.Engine)
+ }()
+
+ select {
+ case <-secondStarted:
+ cancel()
+ case <-time.After(5 * time.Second):
+ t.Fatal("second artifact download did not start")
+ }
+ select {
+ case err := <-done:
+ if !errors.Is(err, context.Canceled) {
+ t.Fatalf("install error = %v, want context cancellation", err)
+ }
+ case <-time.After(5 * time.Second):
+ t.Fatal("cancelled install did not return")
+ }
+ progress.assertSingleTerminal(t, "failed")
+ assertNoArtifactTemps(t, manifest.Engine)
+}
+
+func pinnedArtifact(name, url string, payload []byte) InstallArtifact {
+ sum := sha256.Sum256(payload)
+ return InstallArtifact{Name: name, URL: url, SHA256: hex.EncodeToString(sum[:])}
+}
+
+func artifactInstallManifest(t *testing.T, engine, marker string, artifacts []InstallArtifact) *Manifest {
+ t.Helper()
+ manifest := testEngineManifest(fakeEngineBin)
+ manifest.Engine = engine
+ manifest.DisplayName = "Artifact Test"
+ platform := manifest.Platforms[hostKey()]
+ platform.Detect = []string{marker}
+ platform.Install = &Install{
+ Artifacts: artifacts,
+ Run: []string{fakeEngineBin, "captureargs", marker, "{download_server}", "{download_cudart}"},
+ }
+ manifest.Platforms[hostKey()] = platform
+ if err := manifest.Validate(); err != nil {
+ t.Fatalf("validate fixture: %v", err)
+ }
+ return manifest
+}
+
+func assertNoArtifactTemps(t *testing.T, engine string) {
+ t.Helper()
+ matches, err := filepath.Glob(filepath.Join(os.TempDir(), "nvpair-engine-"+engine+"-*"))
+ if err != nil {
+ t.Fatalf("glob temporary artifacts: %v", err)
+ }
+ if len(matches) != 0 {
+ t.Fatalf("temporary artifacts remain: %v", matches)
+ }
+}
diff --git a/services/nvpair-engine-manager/install_linux_test.go b/services/nvpair-engine-manager/install_linux_test.go
new file mode 100644
index 00000000..b4a88ac5
--- /dev/null
+++ b/services/nvpair-engine-manager/install_linux_test.go
@@ -0,0 +1,108 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+//go:build linux
+
+package main
+
+import (
+ "archive/tar"
+ "bytes"
+ "compress/gzip"
+ "context"
+ "net/http"
+ "net/http/httptest"
+ "os"
+ "path/filepath"
+ "testing"
+ "time"
+)
+
+func TestLlamaCPPInstallPreservesShellArtifactArguments(t *testing.T) {
+ // Both archives wrap their contents in one top-level directory, as every
+ // llama.cpp release tarball does, and the two wrappers are named
+ // differently — the server's after the build tag, the CUDA runtime's after
+ // the whole artifact. The install strips one component from each so the
+ // binaries and their libraries land side by side in the install directory,
+ // which is where detect and runtime.bin look.
+ serverArchive := testTarGZIP(t, "llama-b11146/llama-server", "server")
+ cudartArchive := testTarGZIP(t, "cudart-llama-b11146-bin-ubuntu-cuda-12.8-x64/libcudart.so.12", "runtime")
+ archives := map[string][]byte{
+ "/server.tar.gz": serverArchive,
+ "/cudart.tar.gz": cudartArchive,
+ }
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ payload, ok := archives[r.URL.Path]
+ if !ok {
+ http.NotFound(w, r)
+ return
+ }
+ _, _ = w.Write(payload)
+ }))
+ defer server.Close()
+
+ registry := loadWithOverrides(t, t.TempDir())
+ manifest, ok := registry.Get("llamacpp")
+ if !ok {
+ t.Fatal("llamacpp manifest not loaded")
+ }
+ platform, ok := manifest.Platforms[hostKey()]
+ if !ok || platform.Install == nil {
+ t.Fatalf("llamacpp install is unavailable for %s", hostKey())
+ }
+ platform.Install.Artifacts = []InstallArtifact{
+ pinnedArtifact("server", server.URL+"/server.tar.gz", serverArchive),
+ pinnedArtifact("cudart", server.URL+"/cudart.tar.gz", cudartArchive),
+ }
+ platform.Runtime.Port = 0
+ manifest.Platforms[hostKey()] = platform
+ if err := manifest.Validate(); err != nil {
+ t.Fatalf("validate llama.cpp test manifest: %v", err)
+ }
+
+ baseDir := filepath.Join(t.TempDir(), "Nvidia Corporation", "Personal AI Router", "engine-bin")
+ executor := NewExecutor(registry, NewReporter(nil), nil, baseDir)
+ executor.detectTimeout = 2 * time.Second
+ if err := executor.Install(context.Background(), "llamacpp"); err != nil {
+ t.Fatalf("install llama.cpp with shell artifact arguments: %v", err)
+ }
+
+ for name, want := range map[string]string{
+ "llama-server": "server",
+ "libcudart.so.12": "runtime",
+ } {
+ data, err := os.ReadFile(filepath.Join(baseDir, "llamacpp", name))
+ if err != nil {
+ t.Fatalf("read extracted %s: %v", name, err)
+ }
+ if string(data) != want {
+ t.Errorf("%s contents = %q, want %q", name, data, want)
+ }
+ }
+ assertNoArtifactTemps(t, manifest.Engine)
+}
+
+func testTarGZIP(t *testing.T, name, contents string) []byte {
+ t.Helper()
+ var buffer bytes.Buffer
+ compressor := gzip.NewWriter(&buffer)
+ archive := tar.NewWriter(compressor)
+ data := []byte(contents)
+ if err := archive.WriteHeader(&tar.Header{
+ Name: name,
+ Mode: 0o755,
+ Size: int64(len(data)),
+ }); err != nil {
+ t.Fatalf("write tar header for %s: %v", name, err)
+ }
+ if _, err := archive.Write(data); err != nil {
+ t.Fatalf("write tar contents for %s: %v", name, err)
+ }
+ if err := archive.Close(); err != nil {
+ t.Fatalf("close tar archive for %s: %v", name, err)
+ }
+ if err := compressor.Close(); err != nil {
+ t.Fatalf("close gzip stream for %s: %v", name, err)
+ }
+ return buffer.Bytes()
+}
diff --git a/services/nvpair-engine-manager/install_windows_test.go b/services/nvpair-engine-manager/install_windows_test.go
new file mode 100644
index 00000000..613dce88
--- /dev/null
+++ b/services/nvpair-engine-manager/install_windows_test.go
@@ -0,0 +1,92 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+//go:build windows
+
+package main
+
+import (
+ "archive/zip"
+ "bytes"
+ "context"
+ "net/http"
+ "net/http/httptest"
+ "os"
+ "path/filepath"
+ "testing"
+ "time"
+)
+
+func TestLlamaCPPInstallPreservesDestinationPathWithSpaces(t *testing.T) {
+ serverArchive := testZIP(t, "llama-server.exe", "server")
+ cudartArchive := testZIP(t, "cudart64_12.dll", "runtime")
+ archives := map[string][]byte{
+ "/server.zip": serverArchive,
+ "/cudart.zip": cudartArchive,
+ }
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ payload, ok := archives[r.URL.Path]
+ if !ok {
+ http.NotFound(w, r)
+ return
+ }
+ _, _ = w.Write(payload)
+ }))
+ defer server.Close()
+
+ registry := loadWithOverrides(t, t.TempDir())
+ manifest, ok := registry.Get("llamacpp")
+ if !ok {
+ t.Fatal("llamacpp manifest not loaded")
+ }
+ platform, ok := manifest.Platforms[hostKey()]
+ if !ok || platform.Install == nil {
+ t.Fatalf("llamacpp install is unavailable for %s", hostKey())
+ }
+ platform.Install.Artifacts = []InstallArtifact{
+ pinnedArtifact("server", server.URL+"/server.zip", serverArchive),
+ pinnedArtifact("cudart", server.URL+"/cudart.zip", cudartArchive),
+ }
+ platform.Runtime.Port = 0
+ manifest.Platforms[hostKey()] = platform
+ if err := manifest.Validate(); err != nil {
+ t.Fatalf("validate llama.cpp test manifest: %v", err)
+ }
+
+ baseDir := filepath.Join(t.TempDir(), "Nvidia Corporation", "Personal AI Router", "engine-bin")
+ executor := NewExecutor(registry, NewReporter(nil), nil, baseDir)
+ executor.detectTimeout = 2 * time.Second
+ if err := executor.Install(context.Background(), "llamacpp"); err != nil {
+ t.Fatalf("install llama.cpp into a spaced path: %v", err)
+ }
+
+ for name, want := range map[string]string{
+ "llama-server.exe": "server",
+ "cudart64_12.dll": "runtime",
+ } {
+ data, err := os.ReadFile(filepath.Join(baseDir, "llamacpp", name))
+ if err != nil {
+ t.Fatalf("read extracted %s: %v", name, err)
+ }
+ if string(data) != want {
+ t.Errorf("%s contents = %q, want %q", name, data, want)
+ }
+ }
+}
+
+func testZIP(t *testing.T, name, contents string) []byte {
+ t.Helper()
+ var buffer bytes.Buffer
+ archive := zip.NewWriter(&buffer)
+ file, err := archive.Create(name)
+ if err != nil {
+ t.Fatalf("create zip entry %s: %v", name, err)
+ }
+ if _, err := file.Write([]byte(contents)); err != nil {
+ t.Fatalf("write zip entry %s: %v", name, err)
+ }
+ if err := archive.Close(); err != nil {
+ t.Fatalf("close zip archive: %v", err)
+ }
+ return buffer.Bytes()
+}
diff --git a/services/nvpair-engine-manager/installlayout_test.go b/services/nvpair-engine-manager/installlayout_test.go
new file mode 100644
index 00000000..7b539b2b
--- /dev/null
+++ b/services/nvpair-engine-manager/installlayout_test.go
@@ -0,0 +1,163 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package main
+
+import (
+ "os"
+ "os/exec"
+ "path/filepath"
+ "runtime"
+ "strings"
+ "testing"
+)
+
+// bundledManifestSet loads the compiled-in manifests the way main.go does.
+func bundledManifestSet(t *testing.T) map[string]*Manifest {
+ t.Helper()
+ reg := NewRegistry()
+ if err := reg.LoadFS(bundledManifests, "manifests"); err != nil {
+ t.Fatalf("load bundled manifests: %v", err)
+ }
+ out := map[string]*Manifest{}
+ for _, name := range reg.Names() {
+ manifest, ok := reg.Get(name)
+ if !ok {
+ t.Fatalf("registry lost manifest %q", name)
+ }
+ out[name] = manifest
+ }
+ if len(out) == 0 {
+ t.Fatal("no bundled manifests loaded")
+ }
+ return out
+}
+
+// TestBundledRuntimeBinIsDetected keeps detection and launch pointing at the
+// same file.
+//
+// A detect path is a claim about what the install produces, and nothing checks
+// it at install time: the runner extracts, looks for the declared path, finds
+// nothing, and reports "was not detected after install" with no way to say
+// why. That is what llama.cpp did on macOS and Linux, where the manifest named
+// the layout of a local cmake build (build/bin/llama-server) rather than of the
+// published release archive.
+//
+// Whether a path matches the archive is only knowable from the archive, so the
+// real check is TestLiveBundledInstallLayout, which downloads each one. What is
+// checkable here is that the two paths agree: detecting one file and launching
+// another lets an engine report installed and then fail to start.
+//
+// There is deliberately no rule about how deep a detect path may be. Archive
+// shapes are the vendor's choice and they differ — llama.cpp wraps everything
+// in a build-tagged directory the install has to strip, while Ollama's Linux
+// archive is already bin/ and lib/ and must not be stripped. A convention
+// asserted here would only encode one vendor's habit as if it were a contract.
+func TestBundledRuntimeBinIsDetected(t *testing.T) {
+ for name, manifest := range bundledManifestSet(t) {
+ for key, platform := range manifest.Platforms {
+ bin := platform.Runtime.Bin
+ if bin == "" || len(platform.Detect) == 0 {
+ continue
+ }
+ found := false
+ for _, candidate := range platform.Detect {
+ if candidate == bin {
+ found = true
+ break
+ }
+ }
+ if !found {
+ t.Errorf("%s/%s: runtime.bin %q is not among the detect paths %v — the engine would report installed and then fail to start",
+ name, key, bin, platform.Detect)
+ }
+ }
+ }
+}
+
+// TestBundledInstallLayout downloads every archive the bundled manifests
+// install from, extracts it the way the manifest says to, and checks the detect
+// path appears. This is the check that was missing when llama.cpp shipped a
+// detect path no release archive could satisfy.
+//
+// Gated on an environment variable rather than a build tag: the archives run to
+// gigabytes, but the `live` tag does not currently compile in this package, and
+// a verification nobody can run is not one.
+//
+// Extraction runs through the host's tar, which libarchive-backed tar handles
+// for .tar.gz, .tar.zst and .zip alike, so one host can verify another
+// platform's archive. The --strip-components the manifest declares is applied,
+// because that flag is what decides whether the detect path resolves. A
+// platform whose install is not a tar invocation (Windows llama.cpp uses
+// Expand-Archive) is still covered: only the extraction mechanism differs, and
+// the archive it reads is the one fetched here.
+//
+// NVPAIR_LIVE_LAYOUT=1 go test -run TestBundledInstallLayout -v -timeout 3600s
+func TestBundledInstallLayout(t *testing.T) {
+ if os.Getenv("NVPAIR_LIVE_LAYOUT") == "" {
+ t.Skip("set NVPAIR_LIVE_LAYOUT=1 to download every engine archive and verify its layout")
+ }
+ for name, manifest := range bundledManifestSet(t) {
+ for key, platform := range manifest.Platforms {
+ urls := archiveURLs(platform)
+ if len(urls) == 0 {
+ continue // vendor script or detect-only engine; nothing to unpack
+ }
+ t.Run(name+"/"+key, func(t *testing.T) {
+ installDir := t.TempDir()
+ strip := strings.Contains(strings.Join(platform.Install.Run, " "), "--strip-components=1")
+ for _, url := range urls {
+ extractArchive(t, url, installDir, strip)
+ }
+ for _, candidate := range platform.Detect {
+ if !strings.HasPrefix(candidate, "{install_dir}") {
+ continue
+ }
+ relative := strings.TrimLeft(strings.TrimPrefix(candidate, "{install_dir}"), `/\`)
+ // Manifests spell Windows paths with backslashes; the host
+ // separator is what the extracted tree uses.
+ relative = filepath.FromSlash(strings.ReplaceAll(relative, `\`, "/"))
+ if _, err := os.Stat(filepath.Join(installDir, relative)); err != nil {
+ t.Errorf("detect path %q is absent after extracting %v (strip=%v): %v",
+ candidate, urls, strip, err)
+ }
+ }
+ })
+ }
+ }
+}
+
+// archiveURLs lists the downloads a platform's install unpacks, in the order
+// the manifest extracts them.
+func archiveURLs(platform Platform) []string {
+ if platform.Install == nil || len(platform.Install.Run) == 0 {
+ return nil
+ }
+ var urls []string
+ if platform.Install.Fetch != nil {
+ urls = append(urls, platform.Install.Fetch.URL)
+ }
+ for _, artifact := range platform.Install.Artifacts {
+ urls = append(urls, artifact.URL)
+ }
+ return urls
+}
+
+func extractArchive(t *testing.T, url, installDir string, strip bool) {
+ t.Helper()
+ archive := filepath.Join(t.TempDir(), filepath.Base(url))
+ // curl rather than net/http: these are large, redirected downloads and the
+ // point here is the archive's interior, not the transfer.
+ fetch := exec.Command("curl", "-sSL", "--fail", "--max-time", "900", "-o", archive, url)
+ if out, err := fetch.CombinedOutput(); err != nil {
+ t.Skipf("cannot download %s (%v): %s", url, err, strings.TrimSpace(string(out)))
+ }
+ args := []string{"-xf", archive, "-C", installDir}
+ if strip {
+ args = append(args, "--strip-components=1")
+ }
+ extract := exec.Command("tar", args...)
+ if out, err := extract.CombinedOutput(); err != nil {
+ t.Fatalf("tar %v on %s failed (%v): %s", args, runtime.GOOS, err, strings.TrimSpace(string(out)))
+ }
+}
diff --git a/services/nvpair-engine-manager/launch_controls_test.go b/services/nvpair-engine-manager/launch_controls_test.go
index f8718996..84313046 100644
--- a/services/nvpair-engine-manager/launch_controls_test.go
+++ b/services/nvpair-engine-manager/launch_controls_test.go
@@ -106,10 +106,35 @@ func TestSavedControlsCannotBypassLaunchValidation(t *testing.T) {
}
}
+func TestBundledLlamaCPPDisablesCORSByDefault(t *testing.T) {
+ reg := loadWithOverrides(t, t.TempDir())
+ manifest, ok := reg.Get("llamacpp")
+ if !ok {
+ t.Fatal("llama.cpp manifest not loaded")
+ }
+ for platform, config := range manifest.Platforms {
+ t.Run(platform, func(t *testing.T) {
+ e := settingsExecutor(t, false)
+ settingsState(t, e).plat.Runtime = config.Runtime
+ state, err := e.LaunchSettings("fake")
+ if err != nil {
+ t.Fatal(err)
+ }
+ got, err := launchCORSAssignments(state.LaunchText, config.Runtime.EditableLaunch)
+ if err != nil {
+ t.Fatal(err)
+ }
+ if want := []string{"cors.origins="}; !slices.Equal(got, want) {
+ t.Fatalf("default CORS policy = %v, want %v", got, want)
+ }
+ })
+ }
+}
+
func TestBundledNetworkingControls(t *testing.T) {
reg := loadWithOverrides(t, t.TempDir())
// Adding a bundled engine requires an explicit networking review and cases.
- wantEngines := []string{"lmstudio", "ollama"}
+ wantEngines := []string{"llamacpp", "lmstudio", "ollama"}
names := reg.Names()
slices.Sort(names)
if !slices.Equal(names, wantEngines) {
@@ -126,13 +151,23 @@ func TestBundledNetworkingControls(t *testing.T) {
t.Fatal("missing reviewed networking controls")
}
var valid, invalid []string
- if name == "lmstudio" {
+ switch name {
+ case "lmstudio":
if !reflect.DeepEqual(policy.Controls, []LaunchControl{{Value: "{server.port}", Flags: []string{"--port", "-p"}}, {Value: "{server.host}", Flags: []string{"--bind"}, Env: []string{"LMS_SERVER_HOST"}}, {Value: "{cors.enabled}", Implicit: implicitLaunchValue("true"), Flags: []string{"--cors"}}}) {
t.Fatal("incomplete LM Studio controls")
}
valid = []string{"--port 23456", "--port=23456", "-p 23456", "-p23456", "-p=23456", `"-p" "23456"`, "--port 23456 -p23456", "-- -p23456"}
invalid = []string{"-p0", "-p65536", "-p", "-pno", "--port 23456 -p23457", "-vp23456", "-vp=23456", "--bind 0.0.0.0", "--bind=::", "LMS_SERVER_HOST=0.0.0.0", "--cors=false", "--cors=true", "--cors=", "-- --bind 0.0.0.0"}
- } else {
+ case "llamacpp":
+ if !slices.Equal(policy.FixedArgs, []string{"--sleep-idle-seconds", "300"}) {
+ t.Fatal("llama.cpp idle sleep policy is not fixed")
+ }
+ if !reflect.DeepEqual(policy.Controls, []LaunchControl{{Value: "{server.host}", Flags: []string{"--host"}}, {Value: "{server.port}", Flags: []string{"--port"}}, {Value: "{cors.origins}", Flags: []string{"--cors-origins"}}}) {
+ t.Fatal("incomplete llama.cpp controls")
+ }
+ valid = []string{"--host 127.0.0.1 --port 23456", "--host=127.0.0.1 --port=23456", "--port 23456 --cors-origins https://example.test"}
+ invalid = []string{"--host 0.0.0.0", "--host=::", "--port 0", "--port 65536", "--port", "--cors-origins=*", "--port 23456 --port 23457", "-- --host 0.0.0.0"}
+ default:
if !reflect.DeepEqual(policy.Controls, []LaunchControl{{Value: "{server.host}:{server.port}", Env: []string{"OLLAMA_HOST"}}, {Value: "{cors.origins}", Env: []string{"OLLAMA_ORIGINS"}}}) {
t.Fatal("incomplete Ollama controls")
}
diff --git a/services/nvpair-engine-manager/launch_test.go b/services/nvpair-engine-manager/launch_test.go
index d2b8eb32..ac8e9607 100644
--- a/services/nvpair-engine-manager/launch_test.go
+++ b/services/nvpair-engine-manager/launch_test.go
@@ -22,7 +22,7 @@ func (command launchCommand) text() (string, error) {
func TestResolvedLaunchMatchesBundledEngines(t *testing.T) {
reg := loadWithOverrides(t, t.TempDir())
vars := map[string]string{"host": "127.0.0.1", "port": "12345", "cli": "/test path/lms", "install_dir": "/test path"}
- for _, engine := range []string{"ollama", "lmstudio"} {
+ for _, engine := range []string{"ollama", "lmstudio", "llamacpp"} {
manifest, ok := reg.Get(engine)
if !ok {
t.Fatalf("missing bundled engine %q", engine)
@@ -40,9 +40,19 @@ func TestResolvedLaunchMatchesBundledEngines(t *testing.T) {
if strings.HasPrefix(platformKey, "linux/") {
want = append([]string{"LD_LIBRARY_PATH=/test path/lib/ollama"}, want...)
}
- } else {
+ } else if engine == "lmstudio" {
launch, err = resolveCommandLaunch(platform.Runtime.Start[0], vars)
want = []string{"/test path/lms", "server", "start", "--port", "12345", "--bind", "127.0.0.1"}
+ } else {
+ var bin string
+ bin, err = resolvePlaceholders(platform.Runtime.Bin, vars)
+ if err == nil {
+ launch, err = resolveProcessLaunch(platform.Runtime, bin, vars)
+ }
+ want = []string{"LLAMA_CACHE=/test path-models", bin, "--sleep-idle-seconds", "300", "--host", "127.0.0.1", "--port", "12345", "--cors-origins", ""}
+ if strings.HasPrefix(platformKey, "linux/") {
+ want = append([]string{"LD_LIBRARY_PATH=/test path"}, want...)
+ }
}
if err != nil {
t.Fatal(err)
@@ -97,7 +107,7 @@ func TestLaunchTextReachesChildLiterally(t *testing.T) {
if err != nil {
t.Fatal(err)
}
- t.Cleanup(proc.stop)
+ t.Cleanup(func() { proc.stop(0) })
select {
case <-proc.done:
case <-time.After(10 * time.Second):
diff --git a/services/nvpair-engine-manager/lifecycle.go b/services/nvpair-engine-manager/lifecycle.go
index aa599903..8f9b3271 100644
--- a/services/nvpair-engine-manager/lifecycle.go
+++ b/services/nvpair-engine-manager/lifecycle.go
@@ -5,8 +5,10 @@ package main
import (
"context"
+ "encoding/json"
"errors"
"fmt"
+ "io"
"log/slog"
"net"
"net/http"
@@ -26,6 +28,7 @@ import (
const (
unavailableConfirmations = 3
engineIdentityProbeHeader = "X-NVPAIR-Engine-Identity-Probe"
+ maxProbeJSONBytes = 1 << 20
)
type listenerProbeResult uint8
@@ -279,7 +282,7 @@ func (e *Executor) bringUpProcess(ctx context.Context, st *engineState, engine s
st.mu.Lock()
st.stopping = true
st.mu.Unlock()
- proc.stop()
+ proc.stop(stopGrace(rt))
st.mu.Lock()
st.proc = nil
st.mu.Unlock()
@@ -498,7 +501,7 @@ func (e *Executor) doStop(st *engineState, engine string) error {
return e.reconcileFailedCommandStop(st, engine, rt.Ready == nil || !e.waitUnavailable(rt.Ready, port, time.Second), err)
}
} else if proc != nil {
- proc.stop()
+ proc.stop(stopGrace(rt))
}
e.markStopped(st, engine)
@@ -874,12 +877,30 @@ func (e *Executor) probe(ctx context.Context, p *Probe, port int) bool {
return false
}
- httpcon.DrainAndClose(resp.Body)
want := p.Status
if want == 0 {
want = 200
}
- return resp.StatusCode == want
+ statusMatches := resp.StatusCode == want
+ if p.JSONMatch == nil {
+ httpcon.DrainAndClose(resp.Body)
+ return statusMatches
+ }
+ body, err := io.ReadAll(io.LimitReader(resp.Body, maxProbeJSONBytes+1))
+ httpcon.DrainAndClose(resp.Body)
+ if !statusMatches || err != nil || len(body) > maxProbeJSONBytes {
+ return false
+ }
+ var obj map[string]json.RawMessage
+ if err := json.Unmarshal(body, &obj); err != nil {
+ return false
+ }
+ value, ok := resolveObjectPath(obj, p.JSONMatch.Field)
+ if !ok {
+ return false
+ }
+ var actual string
+ return json.Unmarshal(value, &actual) == nil && actual == p.JSONMatch.Value
}
if p.TCP != "" {
addr, err := resolvePlaceholders(p.TCP, vars)
diff --git a/services/nvpair-engine-manager/lifecycle_stop_test.go b/services/nvpair-engine-manager/lifecycle_stop_test.go
index 7024d68a..3f9a9ef7 100644
--- a/services/nvpair-engine-manager/lifecycle_stop_test.go
+++ b/services/nvpair-engine-manager/lifecycle_stop_test.go
@@ -18,6 +18,26 @@ import (
"time"
)
+func TestStopGrace(t *testing.T) {
+ tests := []struct {
+ name string
+ stop *StopSpec
+ want time.Duration
+ }{
+ {name: "default without stop spec", want: 5 * time.Second},
+ {name: "default with zero grace", stop: &StopSpec{Signal: "term"}, want: 5 * time.Second},
+ {name: "configured grace", stop: &StopSpec{Signal: "term", GraceS: 10}, want: 10 * time.Second},
+ {name: "kill is immediate", stop: &StopSpec{Signal: "kill", GraceS: 10}, want: 0},
+ }
+ for _, tt := range tests {
+ t.Run(tt.name, func(t *testing.T) {
+ if got := stopGrace(Runtime{Stop: tt.stop}); got != tt.want {
+ t.Fatalf("stopGrace() = %s, want %s", got, tt.want)
+ }
+ })
+ }
+}
+
// spawnFakeListener starts a fake-engine copied to binPath, bound to
// 127.0.0.1:port, and skips the test when this host can't resolve the PID/
// image behind a listening port (no lsof/ss, or a /proc-less OS) — the
diff --git a/services/nvpair-engine-manager/lifecycle_stop_unix_test.go b/services/nvpair-engine-manager/lifecycle_stop_unix_test.go
new file mode 100644
index 00000000..a1a2db03
--- /dev/null
+++ b/services/nvpair-engine-manager/lifecycle_stop_unix_test.go
@@ -0,0 +1,101 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+//go:build !windows
+
+package main
+
+import (
+ "path/filepath"
+ "runtime"
+ "testing"
+ "time"
+)
+
+func startTermIgnoringProcess(t *testing.T) *managedProc {
+ t.Helper()
+ ready := filepath.Join(t.TempDir(), "ready")
+ script := `trap '' TERM
+: > "$1"
+while :; do sleep 1; done`
+ proc, err := startManagedProc("/bin/sh", []string{"-c", script, "stubborn-engine", ready}, nil, nil)
+ if err != nil {
+ t.Fatalf("start term-ignoring process: %v", err)
+ }
+ t.Cleanup(func() {
+ select {
+ case <-proc.done:
+ return
+ default:
+ }
+ _ = signalPID(proc.cmd.Process.Pid, true)
+ select {
+ case <-proc.done:
+ case <-time.After(2 * time.Second):
+ t.Errorf("term-ignoring process did not exit during cleanup")
+ }
+ })
+
+ deadline := time.Now().Add(5 * time.Second)
+ for !fileExists(ready) && time.Now().Before(deadline) {
+ time.Sleep(10 * time.Millisecond)
+ }
+ if !fileExists(ready) {
+ t.Fatal("term-ignoring process did not become ready")
+ }
+ return proc
+}
+
+func TestStopHonorsProcessSignalPolicy(t *testing.T) {
+ test := func(t *testing.T, stop StopSpec, minElapsed, maxElapsed time.Duration) {
+ t.Helper()
+ proc := startTermIgnoringProcess(t)
+ manifest := testEngineManifest(fakeEngineBin)
+ key := runtime.GOOS + "/" + runtime.GOARCH
+ platform := manifest.Platforms[key]
+ platform.Runtime.Stop = &stop
+ manifest.Platforms[key] = platform
+ ex := newTestExecutor(t, manifest)
+ state, err := ex.state(manifest.Engine)
+ if err != nil {
+ t.Fatalf("resolve engine state: %v", err)
+ }
+ state.mu.Lock()
+ state.running = true
+ state.healthy = true
+ state.proc = proc
+ state.mu.Unlock()
+
+ started := time.Now()
+ result := make(chan error, 1)
+ go func() {
+ result <- ex.Stop(manifest.Engine)
+ }()
+ select {
+ case err := <-result:
+ if err != nil {
+ t.Fatalf("stop engine: %v", err)
+ }
+ case <-time.After(maxElapsed + time.Second):
+ _ = signalPID(proc.cmd.Process.Pid, true)
+ t.Fatalf("stop did not finish within %s", maxElapsed+time.Second)
+ }
+
+ elapsed := time.Since(started)
+ if elapsed < minElapsed || elapsed > maxElapsed {
+ t.Fatalf("stop elapsed %s, want between %s and %s", elapsed, minElapsed, maxElapsed)
+ }
+ select {
+ case <-proc.done:
+ default:
+ t.Fatal("stop returned before the owned process exited")
+ }
+ }
+
+ t.Run("term escalates after configured grace", func(t *testing.T) {
+ test(t, StopSpec{Signal: "term", GraceS: 1}, 800*time.Millisecond, 3*time.Second)
+ })
+ t.Run("kill bypasses configured grace", func(t *testing.T) {
+ test(t, StopSpec{Signal: "kill", GraceS: 10}, 0, 2*time.Second)
+ })
+}
diff --git a/services/nvpair-engine-manager/manifests/llamacpp.json b/services/nvpair-engine-manager/manifests/llamacpp.json
new file mode 100644
index 00000000..c22b2b0f
--- /dev/null
+++ b/services/nvpair-engine-manager/manifests/llamacpp.json
@@ -0,0 +1,121 @@
+{
+ "$schema": "../manifest.schema.json",
+ "engine": "llamacpp",
+ "display_name": "llama.cpp",
+ "manifest_version": 1,
+ "install": { "mode": "user" },
+ "runtime": {
+ "editable_launch": {
+ "fixed_args": ["--sleep-idle-seconds", "300"],
+ "controls": [
+ { "flags": ["--host"], "value": "{server.host}" },
+ { "flags": ["--port"], "value": "{server.port}" },
+ { "flags": ["--cors-origins"], "value": "{cors.origins}" }
+ ]
+ },
+ "args": ["--sleep-idle-seconds", "300", "--host", "{host}", "--port", "{port}", "--cors-origins", ""],
+ "env": { "LLAMA_CACHE": "{install_dir}-models" },
+ "bind": "127.0.0.1",
+ "port": 8081,
+ "ready": { "http": "http://127.0.0.1:{port}/props", "status": 200, "timeout_s": 60, "json_match": { "field": "role", "value": "router" } },
+ "stop": { "signal": "term", "grace_s": 10 },
+ "health": { "http": "http://127.0.0.1:{port}/health", "status": 200, "interval_s": 5 }
+ },
+ "platforms": {
+ "windows/amd64": {
+ "detect": ["{install_dir}\\llama-server.exe"],
+ "install": {
+ "artifacts": [
+ { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-win-cuda-12.4-x64.zip", "sha256": "3c806a6ceccc3dae1c743ceb1a1fb2cce5b76f40bfbd4c6b7b8afb6ef45a5807" },
+ { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-bin-win-cuda-12.4-x64.zip", "sha256": "8c79a9b226de4b3cacfd1f83d24f962d0773be79f1e7b75c6af4ded7e32ae1d6" }
+ ],
+ "run": ["powershell.exe", "-NoProfile", "-NonInteractive", "-Command", "& { $ErrorActionPreference = 'Stop'; Expand-Archive -LiteralPath $env:NVPAIR_INSTALL_DOWNLOAD_SERVER -DestinationPath $env:NVPAIR_INSTALL_DIR -Force; Expand-Archive -LiteralPath $env:NVPAIR_INSTALL_DOWNLOAD_CUDART -DestinationPath $env:NVPAIR_INSTALL_DIR -Force }"]
+ },
+ "uninstall": { "run": ["cmd", "/c", "rmdir", "/s", "/q", "{install_dir}"] },
+ "runtime": { "bin": "{install_dir}\\llama-server.exe" }
+ },
+ "windows/arm64": {
+ "detect": ["{install_dir}\\llama-server.exe"],
+ "install": {
+ "artifacts": [
+ { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-win-cuda-13.4-arm64.zip", "sha256": "a4060b5031a0e862e225d4a7c4aa403852ebedf598906ef8258a86cb77de8351" },
+ { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-bin-win-cuda-13.4-arm64.zip", "sha256": "642dcde8805b3e3165ca710a5443b3b4044b27d96bd3ee3132473988c9bcb774" }
+ ],
+ "run": ["powershell.exe", "-NoProfile", "-NonInteractive", "-Command", "& { $ErrorActionPreference = 'Stop'; Expand-Archive -LiteralPath $env:NVPAIR_INSTALL_DOWNLOAD_SERVER -DestinationPath $env:NVPAIR_INSTALL_DIR -Force; Expand-Archive -LiteralPath $env:NVPAIR_INSTALL_DOWNLOAD_CUDART -DestinationPath $env:NVPAIR_INSTALL_DIR -Force }"]
+ },
+ "uninstall": { "run": ["cmd", "/c", "rmdir", "/s", "/q", "{install_dir}"] },
+ "runtime": { "bin": "{install_dir}\\llama-server.exe" }
+ },
+ "linux/amd64": {
+ "detect": ["{install_dir}/llama-server"],
+ "install": {
+ "artifacts": [
+ { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-ubuntu-cuda-12.8-x64.tar.gz", "sha256": "c2ab9e19838513ff69d1af8d999ad717dd3c7ee4714ac04c7ed5ab9077c50e4e" },
+ { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-b11146-bin-ubuntu-cuda-12.8-x64.tar.gz", "sha256": "1466daea60aad1144819e151b2bae19d54556cf1da6c129c4f55a5ded2637c25" }
+ ],
+ "run": ["sh", "-c", "tar -xzf \"$1\" -C \"$3\" --strip-components=1 && tar -xzf \"$2\" -C \"$3\" --strip-components=1", "llamacpp-install", "{download_server}", "{download_cudart}", "{install_dir}"]
+ },
+ "uninstall": { "run": ["rm", "-rf", "{install_dir}"] },
+ "runtime": { "bin": "{install_dir}/llama-server", "env": { "LD_LIBRARY_PATH": "{install_dir}" } }
+ },
+ "linux/arm64": {
+ "detect": ["{install_dir}/llama-server"],
+ "install": {
+ "artifacts": [
+ { "name": "server", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-ubuntu-cuda-13.4-arm64.tar.gz", "sha256": "4e00496ab6cdee9c00afb11de3cb9d10f9da7e17147d8ed14ca3af05209b400f" },
+ { "name": "cudart", "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/cudart-llama-b11146-bin-ubuntu-cuda-13.4-arm64.tar.gz", "sha256": "7f46057efcba6338c58ed9f91c06f268c290f3a228fdbdd4bea229dd60b0e094" }
+ ],
+ "run": ["sh", "-c", "tar -xzf \"$1\" -C \"$3\" --strip-components=1 && tar -xzf \"$2\" -C \"$3\" --strip-components=1", "llamacpp-install", "{download_server}", "{download_cudart}", "{install_dir}"]
+ },
+ "uninstall": { "run": ["rm", "-rf", "{install_dir}"] },
+ "runtime": { "bin": "{install_dir}/llama-server", "env": { "LD_LIBRARY_PATH": "{install_dir}" } }
+ },
+ "darwin/arm64": {
+ "detect": ["{install_dir}/llama-server"],
+ "install": {
+ "fetch": { "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-macos-arm64.tar.gz", "sha256": "1ad3f9eff80edb9dbef4259ad564d1720612ef7eea48fa4afed0e54f5f3d5711" },
+ "run": ["tar", "-xzf", "{download}", "-C", "{install_dir}", "--strip-components=1"]
+ },
+ "uninstall": { "run": ["rm", "-rf", "{install_dir}"] },
+ "runtime": { "bin": "{install_dir}/llama-server" }
+ },
+ "darwin/amd64": {
+ "detect": ["{install_dir}/llama-server"],
+ "install": {
+ "fetch": { "url": "https://github.com/ggml-org/llama.cpp/releases/download/b11146/llama-b11146-bin-macos-x64.tar.gz", "sha256": "305f0e3a17d2c01eb205cd0a62128357f1ec3b55329cb084d94e5ec0115d7a3b" },
+ "run": ["tar", "-xzf", "{download}", "-C", "{install_dir}", "--strip-components=1"]
+ },
+ "uninstall": { "run": ["rm", "-rf", "{install_dir}"] },
+ "runtime": { "bin": "{install_dir}/llama-server" }
+ }
+ },
+ "actions": {
+ "list_models": {
+ "description": "List models available in the managed llama.cpp cache.",
+ "http": { "method": "GET", "path": "/models" },
+ "result": { "array": "data", "field": "id" }
+ },
+ "loaded_models": {
+ "description": "List models currently resident in memory.",
+ "http": { "method": "GET", "path": "/models" },
+ "result": { "array": "data", "field": "id", "match": { "field": "status.value", "in": ["loaded"] } }
+ },
+ "pull_model": {
+ "description": "Download a model into the managed cache (params: {\"model\": \"\"}).",
+ "http": { "method": "POST", "path": "/models", "body_schema": { "model": "string" } },
+ "progress_protocol": "llamacpp-models-sse"
+ },
+ "load_model": {
+ "description": "Load a cached model into memory (params: {\"model\": \"\"}).",
+ "http": { "method": "POST", "path": "/models/load", "body_schema": { "model": "string" } }
+ },
+ "unload_model": {
+ "description": "Unload a model from memory (params: {\"model\": \"\"}).",
+ "http": { "method": "POST", "path": "/models/unload", "body_schema": { "model": "string" } }
+ },
+ "delete_model": {
+ "description": "Delete a model from the managed cache (params: {\"model\": \"\"}).",
+ "http": { "method": "DELETE", "path": "/models", "params_in": "query" }
+ }
+ }
+}
diff --git a/services/nvpair-engine-manager/model_test.go b/services/nvpair-engine-manager/model_test.go
index 24d0cdbf..1262840d 100644
--- a/services/nvpair-engine-manager/model_test.go
+++ b/services/nvpair-engine-manager/model_test.go
@@ -17,8 +17,8 @@ import (
//
// This proves the runner correctly drives list/download/use/delete and that
// each step is observable. LM Studio's cmd-shape actions (lms get/ls/load)
-// are covered by TestCmdAction; LM Studio has no CLI model-delete (vendor
-// limitation), which is why only Ollama exposes delete_model.
+// and guarded filesystem deletion are covered separately, as is llama.cpp's
+// query-parameter delete endpoint.
func TestModelLifecycle(t *testing.T) {
m := testEngineManifest(fakeEngineBin) // already has list_models (GET /api/tags)
m.Actions["pull_model"] = Action{HTTP: &ActionHTTP{Method: "POST", Path: "/api/pull"}}
diff --git a/services/nvpair-engine-manager/models.go b/services/nvpair-engine-manager/models.go
index de89064e..31db17aa 100644
--- a/services/nvpair-engine-manager/models.go
+++ b/services/nvpair-engine-manager/models.go
@@ -8,6 +8,7 @@ import (
"context"
"encoding/json"
"log/slog"
+ "strings"
"sync"
"time"
)
@@ -217,13 +218,13 @@ func extractStringsResult(raw json.RawMessage, spec *ActionResult) ([]string, bo
}
// matchRow reports whether an element passes an ActionResult row filter.
-// With Match.In set, Match.Field must decode as a JSON string equal to one of
-// In. With Match.Nonempty set, Match.Field must decode as a JSON array with
-// length > 0 (LM Studio /api/v1/models loaded_instances). A missing or
-// wrong-typed field fails the match, so a row we cannot classify is excluded
-// rather than counted as loaded.
+// Match.Field is a validated dot-separated object path. With Match.In set, the
+// resolved value must decode as a JSON string equal to one of In. With
+// Match.Nonempty set, it must decode as a JSON array with length > 0 (LM Studio
+// /api/v1/models loaded_instances). A missing or wrong-typed path fails the
+// match, so a row we cannot classify is excluded rather than counted as loaded.
func matchRow(el map[string]json.RawMessage, m *ResultMatch) bool {
- fv, ok := el[m.Field]
+ fv, ok := resolveObjectPath(el, m.Field)
if !ok {
return false
}
@@ -245,3 +246,19 @@ func matchRow(el map[string]json.RawMessage, m *ResultMatch) bool {
}
return false
}
+
+func resolveObjectPath(obj map[string]json.RawMessage, path string) (json.RawMessage, bool) {
+ parts := strings.Split(path, ".")
+ value, ok := obj[parts[0]]
+ for _, part := range parts[1:] {
+ if !ok {
+ return nil, false
+ }
+ var nested map[string]json.RawMessage
+ if err := json.Unmarshal(value, &nested); err != nil {
+ return nil, false
+ }
+ value, ok = nested[part]
+ }
+ return value, ok
+}
diff --git a/services/nvpair-engine-manager/models_test.go b/services/nvpair-engine-manager/models_test.go
index a9794f80..96936bee 100644
--- a/services/nvpair-engine-manager/models_test.go
+++ b/services/nvpair-engine-manager/models_test.go
@@ -68,6 +68,12 @@ func TestExtractStrings(t *testing.T) {
spec: &ActionResult{Array: "data", Field: "id", Match: &ResultMatch{Field: "state", In: []string{"loaded"}}},
want: []string{"a"},
},
+ {
+ name: "nested string-in match excludes missing and malformed paths",
+ raw: `{"data":[{"id":"a","status":{"value":"loaded"}},{"id":"b","status":{"value":"not-loaded"}},{"id":"c","status":{}},{"id":"d","status":"loaded"}]}`,
+ spec: &ActionResult{Array: "data", Field: "id", Match: &ResultMatch{Field: "status.value", In: []string{"loaded"}}},
+ want: []string{"a"},
+ },
{
name: "match with no accepted values yields nothing",
raw: `{"data":[{"id":"a","state":"loaded"}]}`,
diff --git a/services/nvpair-engine-manager/proc.go b/services/nvpair-engine-manager/proc.go
index 8c92c773..a1983ec0 100644
--- a/services/nvpair-engine-manager/proc.go
+++ b/services/nvpair-engine-manager/proc.go
@@ -75,21 +75,13 @@ func scanLines(r io.Reader, stream string, onLine func(stream, line string)) {
}
}
-// stop stops the process and waits for it to exit, with no timeout.
+// stop asks the owned process tree to exit, waits for grace, then force-kills
+// it if necessary. A non-positive grace skips the graceful step.
//
-// It sends one platform-appropriate stop signal (see gracefulSignal) and then
-// blocks until the process is gone:
-// - Unix: SIGTERM to the process group — a graceful ask, with no escalation
-// to SIGKILL. A well-behaved engine (Ollama, and the test fake) exits on it.
-// - Windows: taskkill /T /F. Our engines run windowless, and a windowless
-// process can't receive a graceful (non-/F) close, so /F is the only signal
-// that actually stops it — never force-killing there would leave the engine
-// running forever.
-//
-// There is deliberately no timeout: a stop is complete only when the engine has
-// actually exited. On Unix an engine that ignored SIGTERM would not be stopped
-// and this would wait for it; in practice engines exit on SIGTERM.
-func (mp *managedProc) stop() {
+// On Unix the two signals are SIGTERM then SIGKILL. On Windows the engines run
+// windowless, so gracefulSignal is already taskkill /T /F and normally ends the
+// process immediately; the deadline remains a backstop for a failed taskkill.
+func (mp *managedProc) stop(grace time.Duration) {
if mp == nil || mp.cmd == nil || mp.cmd.Process == nil {
return
}
@@ -98,27 +90,41 @@ func (mp *managedProc) stop() {
return // already exited
default:
}
+ if grace <= 0 {
+ _ = signalPID(mp.cmd.Process.Pid, true)
+ <-mp.done
+ return
+ }
_ = gracefulSignal(mp.cmd)
+ timer := time.NewTimer(grace)
+ defer timer.Stop()
+ select {
+ case <-mp.done:
+ return
+ case <-timer.C:
+ }
+ _ = signalPID(mp.cmd.Process.Pid, true)
<-mp.done
}
// terminatePID stops the process with the given PID (and its tree on
// Windows, or its process group on Unix when available): a graceful signal
// first, escalating to a forced kill if the process hasn't exited within
-// grace. It exists to reclaim a PAIR-managed engine orphan adopted on our
-// own port — an instance a prior run spawned and then lost the handle to, so
-// we can only address it by PID rather than through the *exec.Cmd handle
-// managedProc.stop needs. Best-effort: a process that's already gone counts
-// as success. The platform primitives (signalPID, pidAlive) live in
-// proc_windows.go / proc_unix.go.
+// grace; a non-positive grace kills immediately. It exists to reclaim a
+// PAIR-managed engine orphan adopted on our own port — an instance a prior run
+// spawned and then lost the handle to, so we can only address it by PID rather
+// than through the *exec.Cmd handle managedProc.stop needs. Best-effort: a
+// process that's already gone counts as success. The platform primitives
+// (signalPID, pidAlive) live in proc_windows.go / proc_unix.go.
func terminatePID(pid int, grace time.Duration) {
if pid <= 0 {
return
}
- _ = signalPID(pid, false)
if grace <= 0 {
- grace = 5 * time.Second
+ _ = signalPID(pid, true)
+ return
}
+ _ = signalPID(pid, false)
deadline := time.Now().Add(grace)
for time.Now().Before(deadline) {
if !pidAlive(pid) {
diff --git a/services/nvpair-engine-manager/proc_unix.go b/services/nvpair-engine-manager/proc_unix.go
index 876e2b69..12c9b190 100644
--- a/services/nvpair-engine-manager/proc_unix.go
+++ b/services/nvpair-engine-manager/proc_unix.go
@@ -29,10 +29,8 @@ func configureSysProcAttr(cmd *exec.Cmd) {
}
// gracefulSignal sends SIGTERM to the process group (falling back to the
-// process itself). It is the only stop signal engine-manager sends: stop()
-// sends this once and waits for the engine to exit, and never escalates to
-// SIGKILL. A well-behaved engine (Ollama, and the test fake, whose default
-// SIGTERM disposition is to exit) terminates on it.
+// process itself). managedProc.stop escalates to SIGKILL if the group remains
+// alive after the manifest's grace period.
func gracefulSignal(cmd *exec.Cmd) error {
if cmd == nil || cmd.Process == nil {
return nil
@@ -111,9 +109,8 @@ func procImage(pid int) string {
}
// signalPID sends SIGTERM (or SIGKILL when force) to the process group when
-// possible. It is the PID-addressed kill used only by the orphan reclaim (a
-// process we lost the *exec.Cmd handle to), distinct from the normal
-// graceful-only stop() path; force reaches forked helpers (model runners, etc.).
+// possible. The owned-process escalation and orphan reclaim share it so forced
+// stops reach forked helpers (model runners, etc.).
func signalPID(pid int, force bool) error {
if pid <= 0 {
return nil
diff --git a/services/nvpair-engine-manager/proc_windows.go b/services/nvpair-engine-manager/proc_windows.go
index c383c83d..77fd2181 100644
--- a/services/nvpair-engine-manager/proc_windows.go
+++ b/services/nvpair-engine-manager/proc_windows.go
@@ -30,13 +30,11 @@ func configureSysProcAttr(cmd *exec.Cmd) {
}
}
-// gracefulSignal stops the process tree and is the only stop signal
-// engine-manager sends: stop() sends this once and waits for the engine to
-// exit. Windows has no SIGTERM, and the engines we spawn run windowless
-// (CREATE_NO_WINDOW), so a non-/F taskkill only posts WM_CLOSE — which a
-// windowless process can't receive ("can only be terminated forcefully"), i.e.
-// it does nothing. Never force-killing such a process would leave the engine
-// running forever, so on Windows the stop is taskkill /T /F.
+// gracefulSignal stops the process tree. Windows has no SIGTERM, and the
+// engines we spawn run windowless (CREATE_NO_WINDOW), so a non-/F taskkill only
+// posts WM_CLOSE — which a windowless process can't receive ("can only be
+// terminated forcefully"), i.e. it does nothing. The graceful and forced
+// managed-process steps therefore both use taskkill /T /F on Windows.
func gracefulSignal(cmd *exec.Cmd) error {
return taskkill(cmd, true)
}
@@ -172,10 +170,8 @@ func imagePathForPID(pid int) string {
return windows.UTF16ToString(buf[:size])
}
-// signalPID asks the PID's process tree to stop (taskkill /T), escalating to
-// a forced /F kill when force is set. It is the PID-addressed kill used only by
-// the orphan reclaim (a process we lost the *exec.Cmd handle to), distinct from
-// the normal graceful-only stop() path.
+// signalPID asks the PID's process tree to stop (taskkill /T), adding /F when
+// force is set. The owned-process escalation and orphan reclaim share it.
func signalPID(pid int, force bool) error {
if pid <= 0 {
return nil
diff --git a/services/nvpair-engine-manager/pull.go b/services/nvpair-engine-manager/pull.go
index 6ffd8cca..d7642cbe 100644
--- a/services/nvpair-engine-manager/pull.go
+++ b/services/nvpair-engine-manager/pull.go
@@ -14,25 +14,77 @@ package main
// Ollama's /api/pull streams newline-delimited JSON status objects
// ({"status":...,"total":N,"completed":M}); each line maps to a progress event,
// coalesced so only changes in stage/percent are emitted (a single layer streams
-// many byte-progress lines at the same rendered percent). CLI-driven pulls (LM
-// Studio's `lms get`) don't expose structured line progress here, so they emit a
-// single "pulling" marker and return the final result — the security/trust
-// boundary and result contract are identical.
+// many byte-progress lines at the same rendered percent). llama.cpp instead
+// acknowledges POST /models immediately and reports completion on /models/sse;
+// its manifest opts into that named adapter. CLI-driven pulls (LM Studio's
+// `lms get`) emit one "pulling" marker and return the final result.
import (
"bufio"
"bytes"
"context"
"encoding/json"
+ "errors"
"fmt"
"io"
"net/http"
"strconv"
"strings"
+ "time"
)
-// pullModelAction is the manifest action name every engine uses for model pulls.
-const pullModelAction = "pull_model"
+const (
+ // pullModelAction is the manifest action name every engine uses for model pulls.
+ pullModelAction = "pull_model"
+ // pullProgressProtocolLlamaCPPModelsSSE names the pinned llama.cpp router
+ // protocol: subscribe first, POST /models, then await a matching terminal SSE.
+ pullProgressProtocolLlamaCPPModelsSSE = "llamacpp-models-sse"
+)
+
+var errPullProgressTimeout = errors.New("model download made no progress")
+
+type pullProgressWatchdog struct {
+ cancel context.CancelCauseFunc
+ timer *time.Timer
+ timeout time.Duration
+ completedBy map[string]int64
+}
+
+func newPullProgressWatchdog(parent context.Context, timeout time.Duration) (context.Context, *pullProgressWatchdog) {
+ ctx, cancel := context.WithCancelCause(parent)
+ timeoutErr := fmt.Errorf("%w for %s", errPullProgressTimeout, timeout)
+ watchdog := &pullProgressWatchdog{
+ cancel: cancel,
+ timeout: timeout,
+ completedBy: make(map[string]int64),
+ }
+ watchdog.timer = time.AfterFunc(timeout, func() {
+ cancel(timeoutErr)
+ })
+ return ctx, watchdog
+}
+
+func (w *pullProgressWatchdog) stop() {
+ w.timer.Stop()
+ w.cancel(nil)
+}
+
+func (w *pullProgressWatchdog) recordProgress(key string, completed int64) {
+ previous, seen := w.completedBy[key]
+ if completed <= 0 || (seen && completed <= previous) {
+ return
+ }
+ w.completedBy[key] = completed
+ w.timer.Reset(w.timeout)
+}
+
+func (w *pullProgressWatchdog) resolveError(ctx context.Context, fallback error) error {
+ cause := context.Cause(ctx)
+ if errors.Is(cause, errPullProgressTimeout) {
+ return cause
+ }
+ return fallback
+}
// modelFromParams extracts a human-readable model name from an engine:action
// pull_model params object, preferring Ollama's "name" body key then the generic
@@ -70,17 +122,16 @@ func (e *Executor) PullModelStream(ctx context.Context, engine, model string, pa
params, _ = json.Marshal(map[string]string{"name": model, "model": model})
}
- ctx, cancel := context.WithTimeout(ctx, e.actionTimeout)
- defer cancel()
-
// CLI action (e.g. lms get): no structured line progress; emit a start
// marker and return the final result via the existing runner.
if len(act.Cmd) > 0 {
+ actionCtx, cancel := context.WithTimeout(ctx, e.actionTimeout)
+ defer cancel()
st.mu.Lock()
port := st.port
st.mu.Unlock()
e.emitPullProgress(ProgressEvent{Engine: engine, Op: "pull", Stage: "pulling", Message: model})
- return e.runCmdAction(ctx, st, act, port, params)
+ return e.runCmdAction(actionCtx, st, act, port, params)
}
// HTTP action (e.g. Ollama /api/pull): stream NDJSON progress.
@@ -91,19 +142,24 @@ func (e *Executor) PullModelStream(ctx context.Context, engine, model string, pa
if !running {
return nil, fmt.Errorf("engine %q is not running", engine)
}
+ pullCtx, watchdog := newPullProgressWatchdog(ctx, e.pullProgressTimeout)
+ defer watchdog.stop()
+ if act.ProgressProtocol == pullProgressProtocolLlamaCPPModelsSSE {
+ return e.pullModelLlamaCPPSSE(pullCtx, engine, model, act, port, params, watchdog)
+ }
path, err := resolvePlaceholders(act.HTTP.Path, map[string]string{"port": strconv.Itoa(port)})
if err != nil {
return nil, err
}
url := fmt.Sprintf("http://127.0.0.1:%d%s", port, path)
- req, err := http.NewRequestWithContext(ctx, strings.ToUpper(act.HTTP.Method), url, bytes.NewReader(params))
+ req, err := http.NewRequestWithContext(pullCtx, strings.ToUpper(act.HTTP.Method), url, bytes.NewReader(params))
if err != nil {
return nil, err
}
req.Header.Set("Content-Type", "application/json")
resp, err := e.client.Do(req)
if err != nil {
- return nil, fmt.Errorf("pull %q: %w", model, err)
+ return nil, fmt.Errorf("pull %q: %w", model, watchdog.resolveError(pullCtx, err))
}
defer resp.Body.Close()
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
@@ -127,7 +183,9 @@ func (e *Executor) PullModelStream(ctx context.Context, engine, model string, pa
continue
}
last = append(json.RawMessage(nil), line...)
- ev := pullProgressFromLine(engine, line)
+ progress := decodeOllamaPullLine(line)
+ watchdog.recordProgress(progress.progressKey(), progress.Completed)
+ ev := progress.event(engine)
if ev.Stage == lastStage && ev.Percent == lastPct {
continue
}
@@ -135,11 +193,250 @@ func (e *Executor) PullModelStream(ctx context.Context, engine, model string, pa
e.emitPullProgress(ev)
}
if err := sc.Err(); err != nil {
- return nil, fmt.Errorf("pull %q: %w", model, err)
+ return nil, fmt.Errorf("pull %q: %w", model, watchdog.resolveError(pullCtx, err))
}
return last, nil
}
+// pullModelLlamaCPPSSE runs llama.cpp's asynchronous router download protocol.
+// The SSE response must be open before POST /models because terminal events are
+// one-shot broadcasts; subscribing afterward can miss a fast completion.
+func (e *Executor) pullModelLlamaCPPSSE(ctx context.Context, engine, model string, act Action, port int, params json.RawMessage, watchdog *pullProgressWatchdog) (result json.RawMessage, pullErr error) {
+ if strings.TrimSpace(model) == "" {
+ return nil, fmt.Errorf("pull model is required for %s", pullProgressProtocolLlamaCPPModelsSSE)
+ }
+ var requested struct {
+ Model string `json:"model"`
+ }
+ if err := json.Unmarshal(params, &requested); err != nil {
+ return nil, fmt.Errorf("pull %q: decode model params: %w", model, err)
+ }
+ if requested.Model != model {
+ return nil, fmt.Errorf("pull %q: params.model must match the requested model", model)
+ }
+ baseURL := fmt.Sprintf("http://127.0.0.1:%d", port)
+ sseReq, err := http.NewRequestWithContext(ctx, http.MethodGet, baseURL+"/models/sse", nil)
+ if err != nil {
+ return nil, err
+ }
+ sseReq.Header.Set("Accept", "text/event-stream")
+ sseResp, err := e.client.Do(sseReq)
+ if err != nil {
+ return nil, fmt.Errorf("pull %q: subscribe to model progress: %w", model, watchdog.resolveError(ctx, err))
+ }
+ defer sseResp.Body.Close()
+ if sseResp.StatusCode < 200 || sseResp.StatusCode >= 300 {
+ data, _ := io.ReadAll(io.LimitReader(sseResp.Body, 64*1024))
+ return nil, fmt.Errorf("pull %q: progress stream returned HTTP %d: %s", model, sseResp.StatusCode, strings.TrimSpace(string(data)))
+ }
+
+ path, err := resolvePlaceholders(act.HTTP.Path, map[string]string{"port": strconv.Itoa(port)})
+ if err != nil {
+ return nil, err
+ }
+ if cause := context.Cause(ctx); cause != nil {
+ return nil, fmt.Errorf("pull %q: %w", model, cause)
+ }
+ // Cancelling POST /models does not cancel the router's download child. Keep
+ // the bounded handshake alive so cancellation cannot discard its acceptance.
+ startCtx, cancelStart := context.WithTimeout(context.WithoutCancel(ctx), e.pullStartTimeout)
+ defer cancelStart()
+ startReq, err := http.NewRequestWithContext(startCtx, strings.ToUpper(act.HTTP.Method), baseURL+path, bytes.NewReader(params))
+ if err != nil {
+ return nil, err
+ }
+ startReq.Header.Set("Content-Type", "application/json")
+ startResp, err := e.client.Do(startReq)
+ if err != nil {
+ return nil, llamaCPPUnconfirmedStartError(ctx, model, err)
+ }
+ startData, readErr := io.ReadAll(io.LimitReader(startResp.Body, 64*1024))
+ startResp.Body.Close()
+ cancelStart()
+ if readErr != nil {
+ return nil, llamaCPPUnconfirmedStartError(ctx, model, readErr)
+ }
+ if startResp.StatusCode < 200 || startResp.StatusCode >= 300 {
+ return nil, fmt.Errorf("pull %q: engine returned HTTP %d: %s", model, startResp.StatusCode, strings.TrimSpace(string(startData)))
+ }
+ var started struct {
+ Success *bool `json:"success"`
+ }
+ if err := json.Unmarshal(startData, &started); err != nil {
+ return nil, llamaCPPUnconfirmedStartError(ctx, model, err)
+ }
+ if started.Success == nil {
+ return nil, llamaCPPUnconfirmedStartError(ctx, model, errors.New("start response has no success flag"))
+ }
+ if !*started.Success {
+ return nil, fmt.Errorf("pull %q: engine did not accept the download", model)
+ }
+ terminal := false
+ defer func() {
+ if terminal {
+ return
+ }
+ if cause := context.Cause(ctx); cause != nil {
+ pullErr = fmt.Errorf("pull %q: %w", model, cause)
+ }
+ cleanupCtx, cancelCleanup := context.WithTimeout(context.WithoutCancel(ctx), e.pullCleanupTimeout)
+ defer cancelCleanup()
+ if err := e.stopLlamaCPPDownload(cleanupCtx, baseURL, model); err != nil {
+ pullErr = errors.Join(pullErr, fmt.Errorf("pull %q: could not confirm download stopped: %w", model, err))
+ }
+ }()
+ if cause := context.Cause(ctx); cause != nil {
+ return nil, fmt.Errorf("pull %q: %w", model, cause)
+ }
+
+ lastPct := -1
+ sc := bufio.NewScanner(sseResp.Body)
+ sc.Buffer(make([]byte, 0, 64*1024), 1<<20)
+ for sc.Scan() {
+ line := bytes.TrimSpace(sc.Bytes())
+ if !bytes.HasPrefix(line, []byte("data:")) {
+ continue
+ }
+ var event llamaCPPModelsEvent
+ if err := json.Unmarshal(bytes.TrimSpace(bytes.TrimPrefix(line, []byte("data:"))), &event); err != nil || event.Model != model {
+ continue
+ }
+ switch event.Event {
+ case "download_progress":
+ event.recordProgress(watchdog)
+ pct := event.percent()
+ if pct != lastPct {
+ lastPct = pct
+ e.emitPullProgress(ProgressEvent{Engine: engine, Op: "pull", Stage: "downloading", Percent: pct, Message: model})
+ }
+ case "download_finished":
+ terminal = true
+ e.emitPullProgress(ProgressEvent{Engine: engine, Op: "pull", Stage: "success", Percent: 100, Message: model})
+ return json.RawMessage(startData), nil
+ case "download_failed":
+ terminal = true
+ return nil, fmt.Errorf("pull %q: llama.cpp reported download failure", model)
+ }
+ }
+ if err := sc.Err(); err != nil {
+ return nil, fmt.Errorf("pull %q: progress stream: %w", model, watchdog.resolveError(ctx, err))
+ }
+ return nil, fmt.Errorf("pull %q: progress stream ended before completion", model)
+}
+
+// A lost acknowledgement gives us no ownership of a router download: another
+// caller may already be downloading this ID. Preserve the cause without blindly
+// unloading somebody else's model.
+func llamaCPPUnconfirmedStartError(ctx context.Context, model string, err error) error {
+ return fmt.Errorf("pull %q: download acceptance and cancellation could not be confirmed: %w", model, errors.Join(context.Cause(ctx), err))
+}
+
+// stopLlamaCPPDownload uses the same router as the start request. Inventory is
+// checked first because a missed terminal SSE may mean the model is now loaded;
+// unload would then interrupt inference rather than stop an active download.
+func (e *Executor) stopLlamaCPPDownload(ctx context.Context, baseURL, model string) error {
+ req, err := http.NewRequestWithContext(ctx, http.MethodGet, baseURL+"/models", nil)
+ if err != nil {
+ return err
+ }
+ resp, err := e.client.Do(req)
+ if err != nil {
+ return fmt.Errorf("check download inventory: %w", err)
+ }
+ var inventory struct {
+ Data *[]struct {
+ ID string `json:"id"`
+ Status struct {
+ Value string `json:"value"`
+ } `json:"status"`
+ } `json:"data"`
+ }
+ decodeErr := json.NewDecoder(io.LimitReader(resp.Body, 8*1024*1024)).Decode(&inventory)
+ resp.Body.Close()
+ if resp.StatusCode < 200 || resp.StatusCode >= 300 {
+ return fmt.Errorf("check download inventory: HTTP %d", resp.StatusCode)
+ }
+ if decodeErr != nil {
+ return fmt.Errorf("decode download inventory: %w", decodeErr)
+ }
+ if inventory.Data == nil {
+ return errors.New("download inventory has no data array")
+ }
+ for _, entry := range *inventory.Data {
+ if entry.ID != model {
+ continue
+ }
+ if entry.Status.Value == "" {
+ return errors.New("download inventory has no model status")
+ }
+ if entry.Status.Value != "downloading" {
+ return nil
+ }
+ params, err := json.Marshal(map[string]string{"model": model})
+ if err != nil {
+ return err
+ }
+ req, err := http.NewRequestWithContext(ctx, http.MethodPost, baseURL+"/models/unload", bytes.NewReader(params))
+ if err != nil {
+ return err
+ }
+ req.Header.Set("Content-Type", "application/json")
+ resp, err := e.client.Do(req)
+ if err != nil {
+ return fmt.Errorf("stop download: %w", err)
+ }
+ defer resp.Body.Close()
+ if resp.StatusCode < 200 || resp.StatusCode >= 300 {
+ return fmt.Errorf("stop download: HTTP %d", resp.StatusCode)
+ }
+ var stopped struct {
+ Success bool `json:"success"`
+ }
+ if err := json.NewDecoder(io.LimitReader(resp.Body, 64*1024)).Decode(&stopped); err != nil {
+ return fmt.Errorf("decode download stop response: %w", err)
+ }
+ if !stopped.Success {
+ return errors.New("engine did not confirm download stopped")
+ }
+ return nil
+ }
+ return nil
+}
+
+type llamaCPPModelsEvent struct {
+ Model string `json:"model"`
+ Event string `json:"event"`
+ Data struct {
+ Progress map[string]struct {
+ Done int64 `json:"done"`
+ Total int64 `json:"total"`
+ } `json:"progress"`
+ } `json:"data"`
+}
+
+func (e llamaCPPModelsEvent) recordProgress(watchdog *pullProgressWatchdog) {
+ for file, progress := range e.Data.Progress {
+ watchdog.recordProgress(file, progress.Done)
+ }
+}
+
+func (e llamaCPPModelsEvent) percent() int {
+ var done, total int64
+ for _, file := range e.Data.Progress {
+ if file.Total <= 0 {
+ continue
+ }
+ total += file.Total
+ if file.Done > 0 {
+ done += min(file.Done, file.Total)
+ }
+ }
+ if total == 0 {
+ return 0
+ }
+ return int(done * 100 / total)
+}
+
// engineDisplayName returns the manifest display name for user-facing copy.
func (e *Executor) engineDisplayName(engine string) string {
st, err := e.state(engine)
@@ -160,18 +457,36 @@ func (e *Executor) reportPullFailed(engine, model string, err error) string {
return msg
}
-// pullProgressFromLine maps an Ollama /api/pull status line into a ProgressEvent,
-// computing a percentage when the line carries total/completed byte counts.
-func pullProgressFromLine(engine string, line []byte) ProgressEvent {
- var p struct {
- Status string `json:"status"`
- Total int64 `json:"total"`
- Completed int64 `json:"completed"`
+type ollamaPullLine struct {
+ Status string `json:"status"`
+ Digest string `json:"digest"`
+ Total int64 `json:"total"`
+ Completed int64 `json:"completed"`
+}
+
+func decodeOllamaPullLine(line []byte) ollamaPullLine {
+ var progress ollamaPullLine
+ _ = json.Unmarshal(line, &progress)
+ return progress
+}
+
+func (p ollamaPullLine) progressKey() string {
+ if p.Digest != "" {
+ return p.Digest
}
- _ = json.Unmarshal(line, &p)
+ return p.Status
+}
+
+func (p ollamaPullLine) event(engine string) ProgressEvent {
pct := 0
if p.Total > 0 {
pct = int(p.Completed * 100 / p.Total)
}
return ProgressEvent{Engine: engine, Op: "pull", Stage: p.Status, Percent: pct, Message: p.Status}
}
+
+// pullProgressFromLine maps an Ollama /api/pull status line into a ProgressEvent,
+// computing a percentage when the line carries total/completed byte counts.
+func pullProgressFromLine(engine string, line []byte) ProgressEvent {
+ return decodeOllamaPullLine(line).event(engine)
+}
diff --git a/services/nvpair-engine-manager/pull_llamacpp_remote_test.go b/services/nvpair-engine-manager/pull_llamacpp_remote_test.go
new file mode 100644
index 00000000..9c6bbaca
--- /dev/null
+++ b/services/nvpair-engine-manager/pull_llamacpp_remote_test.go
@@ -0,0 +1,138 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package main
+
+import (
+ "bytes"
+ "context"
+ "encoding/json"
+ "fmt"
+ "net/http"
+ "net/http/httptest"
+ "path/filepath"
+ "strconv"
+ "testing"
+ "time"
+
+ "nvpair-shared/clustertrust"
+)
+
+func TestLlamaCPPPullRemoteDisconnectStopsDownload(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ stopped := make(chan struct{})
+ f.unload = func(w http.ResponseWriter, _ *http.Request) {
+ f.downloading.Store(false)
+ if _, err := fmt.Fprint(w, `{"success":true}`); err != nil {
+ t.Errorf("write stop response: %v", err)
+ }
+ close(stopped)
+ }
+ s := &controlServer{exec: f.ex}
+ server := httptest.NewServer(http.HandlerFunc(s.handlePull))
+ t.Cleanup(server.Close)
+ requestLlamaCPPPullAndDisconnect(t, server.Client(), server.URL, f.started)
+ waitLlamaCPPPullSignal(t, stopped)
+ if f.downloading.Load() || f.unloads.Load() != 1 {
+ t.Fatalf("downloading=%t unloads=%d, want remote disconnect to stop download", f.downloading.Load(), f.unloads.Load())
+ }
+}
+
+// Exercise the real ec listener, pinned mTLS, request cancellation, and cleanup
+// in the compiled manager, without a real engine or model download.
+func TestE2ELlamaCPPPullRemoteDisconnectStopsDownload(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ stopped := make(chan struct{})
+ f.unload = func(w http.ResponseWriter, _ *http.Request) {
+ f.downloading.Store(false)
+ if _, err := fmt.Fprint(w, `{"success":true}`); err != nil {
+ t.Errorf("write stop response: %v", err)
+ }
+ close(stopped)
+ }
+ serverCert, serverKey := mintLeaf(t, "pull-server")
+ clientCert, clientKey := mintLeaf(t, "pull-client")
+ serverDir := clusterDirFor(t, serverCert, serverKey, map[string][]byte{"pull-client": clientCert})
+ clientDir := clusterDirFor(t, clientCert, clientKey, map[string][]byte{"pull-server": serverCert})
+ controlPort, err := freePort()
+ if err != nil {
+ t.Fatalf("allocate ec port: %v", err)
+ }
+ state, err := f.ex.state("fake")
+ if err != nil {
+ t.Fatalf("resolve fake router: %v", err)
+ }
+ manifest := testEngineManifest(fakeEngineBin)
+ manifest.Actions[pullModelAction] = Action{
+ HTTP: &ActionHTTP{Method: http.MethodPost, Path: "/models"},
+ ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE,
+ }
+ for key, platform := range manifest.Platforms {
+ platform.Runtime.Port = state.port
+ platform.Runtime.Ready.HTTP = "http://127.0.0.1:{port}/health"
+ platform.Runtime.Health = nil
+ manifest.Platforms[key] = platform
+ }
+ cfg, home := t.TempDir(), t.TempDir()
+ for _, dir := range []string{
+ filepath.Join(cfg, "Nvidia Corporation", "Personal AI Router", "engines"),
+ filepath.Join(home, "Library", "Application Support", "Nvidia Corporation", "Personal AI Router", "engines"),
+ } {
+ writeE2EManifest(t, dir, manifest)
+ }
+ manager := startE2EManager(t, cfg, home, "--control-port", strconv.Itoa(controlPort), "--cluster-dir", serverDir, "--loaded-poll-interval", "0")
+ // The fake router is already listening. Start adopts it using its readiness
+ // probe; the child never spawns a real engine or another fake listener.
+ send(t, manager.stdin, 1, "engine:start", map[string]string{"engine": "fake"})
+ var status EngineStatus
+ if err := json.Unmarshal(waitResult(t, manager.frames, "1", 10*time.Second), &status); err != nil {
+ t.Fatalf("decode start status: %v", err)
+ }
+ if !status.Running || status.Port != state.port {
+ t.Fatalf("engine status = %+v, want running fake router at %d", status, state.port)
+ }
+ waitPortServing(t, controlPort)
+ tlsConfig, ok := clustertrust.Open(clientDir).ClientTLSConfig("pull-server")
+ if !ok {
+ t.Fatal("client could not resolve pinned server TLS configuration")
+ }
+ transport := &http.Transport{TLSClientConfig: tlsConfig}
+ t.Cleanup(transport.CloseIdleConnections)
+ client := &http.Client{Transport: transport}
+ requestLlamaCPPPullAndDisconnect(t, client, fmt.Sprintf("https://127.0.0.1:%d", controlPort), f.started)
+ waitLlamaCPPPullSignal(t, stopped)
+ if f.downloading.Load() || f.unloads.Load() != 1 {
+ t.Fatalf("downloading=%t unloads=%d, want compiled manager to stop download", f.downloading.Load(), f.unloads.Load())
+ }
+ // Shut the fixture down first: the adopted listener is owned by this test,
+ // and manager shutdown must not attempt to terminate the test process.
+ f.server.Close()
+ manager.stop(t)
+}
+
+func requestLlamaCPPPullAndDisconnect(t *testing.T, client *http.Client, baseURL string, started <-chan struct{}) {
+ t.Helper()
+ body, err := json.Marshal(pullRequest{OpID: "disconnect-test", Engine: "fake", Model: llamaCPPPullTestModel})
+ if err != nil {
+ t.Fatalf("encode remote pull: %v", err)
+ }
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ req, err := http.NewRequestWithContext(ctx, http.MethodPost, baseURL+controlPullPath, bytes.NewReader(body))
+ if err != nil {
+ t.Fatalf("create remote pull: %v", err)
+ }
+ req.Header.Set("Content-Type", "application/json")
+ resp, err := client.Do(req)
+ if err != nil {
+ t.Fatalf("start remote pull: %v", err)
+ }
+ defer resp.Body.Close()
+ if resp.StatusCode != http.StatusOK {
+ t.Fatalf("remote pull returned HTTP %d", resp.StatusCode)
+ }
+ waitLlamaCPPPullSignal(t, started)
+ if err := resp.Body.Close(); err != nil {
+ t.Fatalf("disconnect remote pull: %v", err)
+ }
+}
diff --git a/services/nvpair-engine-manager/pull_llamacpp_test.go b/services/nvpair-engine-manager/pull_llamacpp_test.go
new file mode 100644
index 00000000..762ecb68
--- /dev/null
+++ b/services/nvpair-engine-manager/pull_llamacpp_test.go
@@ -0,0 +1,499 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package main
+
+import (
+ "context"
+ "encoding/json"
+ "errors"
+ "fmt"
+ "io"
+ "net/http"
+ "net/http/httptest"
+ "strings"
+ "sync/atomic"
+ "testing"
+ "time"
+)
+
+const llamaCPPPullTestModel = "owner/repo:Q4_K_M"
+
+// The download belongs to the router, not to the SSE request. Only unload
+// clears downloading, so disconnecting the stream alone cannot pass a test.
+type llamaCPPPullFixture struct {
+ ex *Executor
+ server *httptest.Server
+ started chan struct{}
+ endStream chan struct{}
+ downloading atomic.Bool
+ unloads atomic.Int32
+ inventories atomic.Int32
+ start http.HandlerFunc
+ inventory http.HandlerFunc
+ unload http.HandlerFunc
+ stream http.HandlerFunc
+}
+
+func newLlamaCPPPullFixture(t *testing.T) *llamaCPPPullFixture {
+ t.Helper()
+ f := &llamaCPPPullFixture{started: make(chan struct{}), endStream: make(chan struct{})}
+ mux := http.NewServeMux()
+ mux.HandleFunc("/health", func(w http.ResponseWriter, _ *http.Request) {
+ w.WriteHeader(http.StatusOK)
+ })
+ mux.HandleFunc("/models/sse", func(w http.ResponseWriter, r *http.Request) {
+ if f.stream != nil {
+ f.stream(w, r)
+ return
+ }
+ w.Header().Set("Content-Type", "text/event-stream")
+ if _, err := fmt.Fprint(w, ": ready\n\n"); err != nil {
+ t.Errorf("write SSE greeting: %v", err)
+ return
+ }
+ if err := http.NewResponseController(w).Flush(); err != nil {
+ t.Errorf("flush SSE greeting: %v", err)
+ return
+ }
+ select {
+ case <-r.Context().Done():
+ case <-f.endStream:
+ }
+ })
+ mux.HandleFunc("/models", func(w http.ResponseWriter, r *http.Request) {
+ switch r.Method {
+ case http.MethodPost:
+ data, err := io.ReadAll(r.Body)
+ if err != nil {
+ t.Errorf("read start request: %v", err)
+ http.Error(w, "invalid body", http.StatusBadRequest)
+ return
+ }
+ var body struct {
+ Model string `json:"model"`
+ }
+ if err := json.Unmarshal(data, &body); err != nil || body.Model != llamaCPPPullTestModel {
+ t.Errorf("decode start request = %+v, error %v", body, err)
+ http.Error(w, "wrong model", http.StatusBadRequest)
+ return
+ }
+ close(f.started)
+ if f.start != nil {
+ f.start(w, r)
+ return
+ }
+ f.downloading.Store(true)
+ if _, err := fmt.Fprint(w, `{"success":true}`); err != nil {
+ t.Errorf("write start response: %v", err)
+ }
+ case http.MethodGet:
+ f.inventories.Add(1)
+ if f.inventory != nil {
+ f.inventory(w, r)
+ return
+ }
+ if _, err := fmt.Fprintf(w, `{"data":[{"id":%q,"status":{"value":"downloading"}},{"id":"other/model","status":{"value":"downloading"}}]}`, llamaCPPPullTestModel); err != nil {
+ t.Errorf("write inventory: %v", err)
+ }
+ default:
+ http.Error(w, "unexpected method", http.StatusMethodNotAllowed)
+ }
+ })
+ mux.HandleFunc("/models/unload", func(w http.ResponseWriter, r *http.Request) {
+ f.unloads.Add(1)
+ var body struct {
+ Model string `json:"model"`
+ }
+ if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
+ t.Errorf("decode stop request: %v", err)
+ http.Error(w, "invalid body", http.StatusBadRequest)
+ return
+ }
+ if r.Method != http.MethodPost || body.Model != llamaCPPPullTestModel {
+ t.Errorf("stop request = %s %+v, want POST for %s", r.Method, body, llamaCPPPullTestModel)
+ http.Error(w, "wrong download", http.StatusBadRequest)
+ return
+ }
+ if f.unload != nil {
+ f.unload(w, r)
+ return
+ }
+ f.downloading.Store(false)
+ if _, err := fmt.Fprint(w, `{"success":true}`); err != nil {
+ t.Errorf("write stop response: %v", err)
+ }
+ })
+ f.server = httptest.NewServer(mux)
+ t.Cleanup(f.server.Close)
+ f.ex = newHTTPPullTestExecutor(t, f.server, Action{
+ HTTP: &ActionHTTP{Method: http.MethodPost, Path: "/models"},
+ ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE,
+ })
+ return f
+}
+
+func (f *llamaCPPPullFixture) pull(ctx context.Context) error {
+ _, err := f.ex.PullModelStream(ctx, "fake", llamaCPPPullTestModel, nil)
+ return err
+}
+
+func waitLlamaCPPPullSignal(t *testing.T, signal <-chan struct{}) {
+ t.Helper()
+ select {
+ case <-signal:
+ case <-time.After(5 * time.Second):
+ t.Fatal("timed out waiting for pull request")
+ }
+}
+
+func TestLlamaCPPPullCancellationStopsDownload(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ ctx, cancel := context.WithCancel(context.Background())
+ defer cancel()
+ go func() {
+ <-f.started
+ cancel()
+ }()
+ if err := f.pull(ctx); !errors.Is(err, context.Canceled) {
+ t.Fatalf("pull error = %v, want cancellation", err)
+ }
+ if f.downloading.Load() || f.unloads.Load() != 1 {
+ t.Fatalf("downloading=%t unloads=%d, want stopped with one unload", f.downloading.Load(), f.unloads.Load())
+ }
+}
+
+func TestLlamaCPPPullInactivityStopsDownload(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ f.ex.pullProgressTimeout = 100 * time.Millisecond
+ if err := f.pull(context.Background()); !errors.Is(err, errPullProgressTimeout) {
+ t.Fatalf("pull error = %v, want inactivity timeout", err)
+ }
+ if f.downloading.Load() || f.unloads.Load() != 1 {
+ t.Fatalf("downloading=%t unloads=%d, want stopped with one unload", f.downloading.Load(), f.unloads.Load())
+ }
+}
+
+func TestLlamaCPPPullStreamEOFStopsDownload(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ go func() {
+ <-f.started
+ close(f.endStream)
+ }()
+ if err := f.pull(context.Background()); err == nil || !strings.Contains(err.Error(), "progress stream ended before completion") {
+ t.Fatalf("pull error = %v, want premature stream EOF", err)
+ }
+ if f.downloading.Load() || f.unloads.Load() != 1 {
+ t.Fatalf("downloading=%t unloads=%d, want stopped with one unload", f.downloading.Load(), f.unloads.Load())
+ }
+}
+
+func TestLlamaCPPPullStreamReadErrorStopsDownload(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ f.stream = func(w http.ResponseWriter, r *http.Request) {
+ w.Header().Set("Content-Length", "1000")
+ if _, err := fmt.Fprint(w, ": ready\n\n"); err != nil {
+ t.Errorf("write SSE greeting: %v", err)
+ return
+ }
+ if err := http.NewResponseController(w).Flush(); err != nil {
+ t.Errorf("flush SSE greeting: %v", err)
+ return
+ }
+ select {
+ case <-f.started:
+ case <-r.Context().Done():
+ }
+ }
+ if err := f.pull(context.Background()); err == nil || !strings.Contains(err.Error(), "unexpected EOF") {
+ t.Fatalf("pull error = %v, want truncated SSE read", err)
+ }
+ if f.downloading.Load() || f.unloads.Load() != 1 {
+ t.Fatalf("downloading=%t unloads=%d, want stopped with one unload", f.downloading.Load(), f.unloads.Load())
+ }
+}
+
+func TestLlamaCPPPullCancellationWaitsForStartAcceptance(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ releaseStart := make(chan struct{}, 1)
+ defer close(releaseStart)
+ f.start = func(w http.ResponseWriter, r *http.Request) {
+ select {
+ case <-releaseStart:
+ case <-r.Context().Done():
+ t.Error("start handshake was cancelled before acceptance")
+ return
+ }
+ f.downloading.Store(true)
+ if _, err := fmt.Fprint(w, `{"success":true}`); err != nil {
+ t.Errorf("write accepted response: %v", err)
+ }
+ }
+ ctx, cancel := context.WithCancel(context.Background())
+ defer cancel()
+ result := make(chan error, 1)
+ go func() { result <- f.pull(ctx) }()
+ waitLlamaCPPPullSignal(t, f.started)
+ cancel()
+ select {
+ case err := <-result:
+ t.Fatalf("pull returned before start acceptance: %v", err)
+ case <-time.After(30 * time.Millisecond):
+ }
+ // Send instead of closing so the deferred close also releases the handler if
+ // an assertion fails before this point.
+ releaseStart <- struct{}{}
+ select {
+ case err := <-result:
+ if !errors.Is(err, context.Canceled) || f.downloading.Load() || f.unloads.Load() != 1 {
+ t.Fatalf("error=%v downloading=%t unloads=%d", err, f.downloading.Load(), f.unloads.Load())
+ }
+ case <-time.After(5 * time.Second):
+ t.Fatal("pull did not return after acceptance and cleanup")
+ }
+}
+
+func TestLlamaCPPPullCancelledBeforeStartDoesNotDownload(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ ctx, cancel := context.WithCancel(context.Background())
+ cancel()
+ if err := f.pull(ctx); !errors.Is(err, context.Canceled) {
+ t.Fatalf("pull error = %v, want cancellation", err)
+ }
+ select {
+ case <-f.started:
+ t.Fatal("cancelled pull sent a start request")
+ default:
+ }
+ if f.unloads.Load() != 0 || f.inventories.Load() != 0 {
+ t.Fatal("cancelled pull attempted cleanup without starting")
+ }
+}
+
+func TestLlamaCPPPullRejectsInvalidModelParamsBeforeStarting(t *testing.T) {
+ for _, tc := range []struct{ name, params string }{
+ {"different model", `{"model":"other/model"}`},
+ {"missing model", `{}`},
+ {"invalid JSON", `{"model":`},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ _, err := f.ex.PullModelStream(context.Background(), "fake", llamaCPPPullTestModel, json.RawMessage(tc.params))
+ if err == nil {
+ t.Fatal("invalid model params were accepted")
+ }
+ select {
+ case <-f.started:
+ t.Fatal("invalid params sent a start request")
+ default:
+ }
+ if f.inventories.Load() != 0 || f.unloads.Load() != 0 {
+ t.Fatal("invalid params triggered cleanup")
+ }
+ })
+ }
+}
+
+func TestLlamaCPPPullRejectedStartPreservesOtherDownload(t *testing.T) {
+ test := func(name string, status int, body string) {
+ t.Run(name, func(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ f.downloading.Store(true)
+ f.start = func(w http.ResponseWriter, _ *http.Request) {
+ w.WriteHeader(status)
+ if _, err := fmt.Fprint(w, body); err != nil {
+ t.Errorf("write rejected start: %v", err)
+ }
+ }
+ if err := f.pull(context.Background()); err == nil {
+ t.Fatal("rejected start returned success")
+ }
+ if !f.downloading.Load() || f.unloads.Load() != 0 || f.inventories.Load() != 0 {
+ t.Fatal("rejected start touched another download")
+ }
+ })
+ }
+ test("HTTP rejection", http.StatusConflict, `{"error":"already exists"}`)
+ test("negative acknowledgement", http.StatusOK, `{"success":false}`)
+}
+
+func TestLlamaCPPPullUnconfirmedStartDoesNotUnload(t *testing.T) {
+ test := func(name string, respond func(*testing.T, http.ResponseWriter, *http.Request)) {
+ t.Run(name, func(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ f.ex.pullStartTimeout = 100 * time.Millisecond
+ f.downloading.Store(true)
+ ctx, cancel := context.WithCancel(context.Background())
+ defer cancel()
+ f.start = func(w http.ResponseWriter, r *http.Request) {
+ cancel()
+ respond(t, w, r)
+ }
+ err := f.pull(ctx)
+ if !errors.Is(err, context.Canceled) || !strings.Contains(err.Error(), "acceptance and cancellation could not be confirmed") {
+ t.Fatalf("pull error = %v, want unconfirmed acceptance preserving cancellation", err)
+ }
+ if !f.downloading.Load() || f.unloads.Load() != 0 || f.inventories.Load() != 0 {
+ t.Fatal("unconfirmed start unloaded an unowned download")
+ }
+ })
+ }
+ test("malformed acknowledgement", func(t *testing.T, w http.ResponseWriter, _ *http.Request) {
+ if _, err := fmt.Fprint(w, "not JSON"); err != nil {
+ t.Errorf("write malformed acknowledgement: %v", err)
+ }
+ })
+ test("missing success flag", func(t *testing.T, w http.ResponseWriter, _ *http.Request) {
+ if _, err := fmt.Fprint(w, `{}`); err != nil {
+ t.Errorf("write incomplete acknowledgement: %v", err)
+ }
+ })
+ test("null success flag", func(t *testing.T, w http.ResponseWriter, _ *http.Request) {
+ if _, err := fmt.Fprint(w, `{"success":null}`); err != nil {
+ t.Errorf("write null acknowledgement: %v", err)
+ }
+ })
+ test("truncated acknowledgement", func(t *testing.T, w http.ResponseWriter, _ *http.Request) {
+ w.Header().Set("Content-Length", "1000")
+ if _, err := fmt.Fprint(w, `{"success":`); err != nil {
+ t.Errorf("write truncated acknowledgement: %v", err)
+ }
+ })
+ test("start header timeout", func(_ *testing.T, _ http.ResponseWriter, r *http.Request) { <-r.Context().Done() })
+ test("start body timeout", func(t *testing.T, w http.ResponseWriter, r *http.Request) {
+ if err := http.NewResponseController(w).Flush(); err != nil {
+ t.Errorf("flush start headers: %v", err)
+ return
+ }
+ <-r.Context().Done()
+ })
+}
+
+func TestLlamaCPPPullTerminalEventsSkipCleanup(t *testing.T) {
+ for _, event := range []string{"download_finished", "download_failed"} {
+ t.Run(event, func(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ f.stream = func(w http.ResponseWriter, r *http.Request) {
+ if err := http.NewResponseController(w).Flush(); err != nil {
+ t.Errorf("flush SSE headers: %v", err)
+ return
+ }
+ select {
+ case <-f.started:
+ case <-r.Context().Done():
+ return
+ }
+ if _, err := fmt.Fprintf(w, "data: {\"model\":%q,\"event\":%q}\n\n", llamaCPPPullTestModel, event); err != nil {
+ t.Errorf("write terminal event: %v", err)
+ }
+ }
+ err := f.pull(context.Background())
+ if (err == nil) != (event == "download_finished") {
+ t.Fatalf("terminal %s returned error %v", event, err)
+ }
+ if f.unloads.Load() != 0 || f.inventories.Load() != 0 {
+ t.Fatal("terminal event triggered cleanup")
+ }
+ })
+ }
+}
+
+func TestLlamaCPPPullCleanupPreservesCompletedModels(t *testing.T) {
+ for _, status := range []string{"downloaded", "loaded", "unloaded", "missing"} {
+ t.Run(status, func(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ close(f.endStream)
+ f.inventory = func(w http.ResponseWriter, _ *http.Request) {
+ f.downloading.Store(false)
+ body := `{"data":[{"id":"other/model","status":{"value":"downloading"}}]}`
+ if status != "missing" {
+ body = fmt.Sprintf(`{"data":[{"id":%q,"status":{"value":%q}}]}`, llamaCPPPullTestModel, status)
+ }
+ if _, err := fmt.Fprint(w, body); err != nil {
+ t.Errorf("write completed inventory: %v", err)
+ }
+ }
+ err := f.pull(context.Background())
+ if err == nil || !strings.Contains(err.Error(), "progress stream ended before completion") || strings.Contains(err.Error(), "could not confirm") {
+ t.Fatalf("pull error = %v, want only premature SSE termination", err)
+ }
+ if f.unloads.Load() != 0 || f.inventories.Load() != 1 {
+ t.Fatalf("inventories=%d unloads=%d, want completed model preserved", f.inventories.Load(), f.unloads.Load())
+ }
+ })
+ }
+}
+
+func TestLlamaCPPPullCleanupFailurePreservesCancellation(t *testing.T) {
+ test := func(name string, status int, body string) {
+ t.Run(name, func(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ ctx, cancel := context.WithCancel(context.Background())
+ defer cancel()
+ f.start = func(w http.ResponseWriter, _ *http.Request) {
+ f.downloading.Store(true)
+ cancel()
+ if _, err := fmt.Fprint(w, `{"success":true}`); err != nil {
+ t.Errorf("write start response: %v", err)
+ }
+ }
+ f.unload = func(w http.ResponseWriter, _ *http.Request) {
+ w.WriteHeader(status)
+ if _, err := fmt.Fprint(w, body); err != nil {
+ t.Errorf("write cleanup failure: %v", err)
+ }
+ }
+ err := f.pull(ctx)
+ if !errors.Is(err, context.Canceled) || !strings.Contains(err.Error(), "could not confirm download stopped") {
+ t.Fatalf("pull error = %v, want cancellation and cleanup failure", err)
+ }
+ if !f.downloading.Load() || f.unloads.Load() != 1 {
+ t.Fatal("failed cleanup was treated as a confirmed stop or retried")
+ }
+ })
+ }
+ test("HTTP rejection", http.StatusServiceUnavailable, "unavailable")
+ test("malformed acknowledgement", http.StatusOK, "invalid JSON")
+ test("negative acknowledgement", http.StatusOK, `{"success":false}`)
+}
+
+func TestLlamaCPPPullCleanupTimeoutPreservesWatchdogCause(t *testing.T) {
+ for _, route := range []string{"inventory", "unload"} {
+ t.Run(route, func(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ f.ex.pullProgressTimeout = 100 * time.Millisecond
+ f.ex.pullCleanupTimeout = 100 * time.Millisecond
+ hang := func(_ http.ResponseWriter, r *http.Request) { <-r.Context().Done() }
+ if route == "inventory" {
+ f.inventory = hang
+ } else {
+ f.unload = hang
+ }
+ err := f.pull(context.Background())
+ if !errors.Is(err, errPullProgressTimeout) || !errors.Is(err, context.DeadlineExceeded) || !strings.Contains(err.Error(), "could not confirm download stopped") {
+ t.Fatalf("pull error = %v, want watchdog and cleanup timeout", err)
+ }
+ if !f.downloading.Load() {
+ t.Fatal("cleanup timeout was treated as a confirmed stop")
+ }
+ })
+ }
+}
+
+func TestLlamaCPPPullInvalidInventoryDoesNotConfirmStop(t *testing.T) {
+ for _, body := range []string{"not JSON", `{}`, `{"data":null}`, `{"data":{}}`, fmt.Sprintf(`{"data":[{"id":%q}]}`, llamaCPPPullTestModel)} {
+ t.Run(body, func(t *testing.T) {
+ f := newLlamaCPPPullFixture(t)
+ close(f.endStream)
+ f.inventory = func(w http.ResponseWriter, _ *http.Request) {
+ if _, err := fmt.Fprint(w, body); err != nil {
+ t.Errorf("write invalid inventory: %v", err)
+ }
+ }
+ err := f.pull(context.Background())
+ if err == nil || !strings.Contains(err.Error(), "could not confirm download stopped") || f.unloads.Load() != 0 {
+ t.Fatalf("error=%v unloads=%d, want failed inventory validation", err, f.unloads.Load())
+ }
+ })
+ }
+}
diff --git a/services/nvpair-engine-manager/pull_test.go b/services/nvpair-engine-manager/pull_test.go
index 76f48197..423e3d10 100644
--- a/services/nvpair-engine-manager/pull_test.go
+++ b/services/nvpair-engine-manager/pull_test.go
@@ -7,10 +7,16 @@ import (
"bytes"
"context"
"encoding/json"
+ "fmt"
+ "net"
+ "net/http"
"net/http/httptest"
+ "strconv"
"strings"
"sync"
+ "sync/atomic"
"testing"
+ "time"
)
func TestPullProgressFromLine(t *testing.T) {
@@ -59,6 +65,325 @@ func TestModelFromParams(t *testing.T) {
}
}
+func TestLlamaCPPModelsEventPercentAggregatesFiles(t *testing.T) {
+ var event llamaCPPModelsEvent
+ err := json.Unmarshal([]byte(`{
+ "model":"owner/repo:Q4_K_M",
+ "event":"download_progress",
+ "data":{"progress":{
+ "model.gguf":{"done":75,"total":100},
+ "mmproj.gguf":{"done":25,"total":100}
+ }}
+ }`), &event)
+ if err != nil {
+ t.Fatalf("decode event: %v", err)
+ }
+ if got := event.percent(); got != 50 {
+ t.Fatalf("aggregate percent = %d, want 50", got)
+ }
+}
+
+func newHTTPPullTestExecutor(t *testing.T, server *httptest.Server, action Action) *Executor {
+ t.Helper()
+ _, portText, err := net.SplitHostPort(server.Listener.Addr().String())
+ if err != nil {
+ t.Fatalf("split test server address: %v", err)
+ }
+ port, err := strconv.Atoi(portText)
+ if err != nil {
+ t.Fatalf("parse test server port: %v", err)
+ }
+ manifest := testEngineManifest(fakeEngineBin)
+ manifest.Actions[pullModelAction] = action
+ registry := NewRegistry()
+ registry.engines[manifest.Engine] = manifest
+ executor := NewExecutor(registry, NewReporter(nil), nil, t.TempDir())
+ state, err := executor.state(manifest.Engine)
+ if err != nil {
+ t.Fatalf("resolve engine state: %v", err)
+ }
+ state.running = true
+ state.port = port
+ return executor
+}
+
+func TestPullModelLlamaCPPSSESubscribesBeforeStarting(t *testing.T) {
+ const model = "owner/repo:Q4_K_M"
+ subscribed := make(chan struct{})
+ started := make(chan struct{})
+ var postBeforeSubscribe atomic.Bool
+ mux := http.NewServeMux()
+ mux.HandleFunc("/models/sse", func(w http.ResponseWriter, r *http.Request) {
+ w.Header().Set("Content-Type", "text/event-stream")
+ close(subscribed)
+ _, _ = fmt.Fprint(w, ": ready\n\n")
+ w.(http.Flusher).Flush()
+ select {
+ case <-started:
+ case <-r.Context().Done():
+ return
+ }
+ write := func(event map[string]any) {
+ data, err := json.Marshal(event)
+ if err != nil {
+ t.Errorf("encode SSE event: %v", err)
+ return
+ }
+ _, _ = fmt.Fprintf(w, "data: %s\n\n", data)
+ w.(http.Flusher).Flush()
+ }
+ write(map[string]any{"model": "other/model", "event": "download_finished", "data": map[string]any{}})
+ write(map[string]any{
+ "model": model,
+ "event": "download_progress",
+ "data": map[string]any{"progress": map[string]any{
+ "model.gguf": map[string]int64{"done": 75, "total": 100},
+ "mmproj.gguf": map[string]int64{"done": 25, "total": 100},
+ }},
+ })
+ write(map[string]any{"model": model, "event": "download_finished", "data": map[string]any{}})
+ })
+ mux.HandleFunc("/models", func(w http.ResponseWriter, r *http.Request) {
+ select {
+ case <-subscribed:
+ default:
+ postBeforeSubscribe.Store(true)
+ }
+ var body struct {
+ Model string `json:"model"`
+ }
+ if err := json.NewDecoder(r.Body).Decode(&body); err != nil || body.Model != model {
+ http.Error(w, "invalid model", http.StatusBadRequest)
+ return
+ }
+ close(started)
+ w.Header().Set("Content-Type", "application/json")
+ _, _ = fmt.Fprint(w, `{"success":true}`)
+ })
+ server := httptest.NewServer(mux)
+ defer server.Close()
+
+ ex := newHTTPPullTestExecutor(t, server, Action{
+ HTTP: &ActionHTTP{Method: http.MethodPost, Path: "/models"},
+ ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE,
+ })
+ progress, cancel := ex.progress.subscribe("fake")
+ defer cancel()
+
+ result, err := ex.PullModelStream(context.Background(), "fake", model, json.RawMessage(`{"model":"`+model+`"}`))
+ if err != nil {
+ t.Fatalf("pull model: %v", err)
+ }
+ var response struct {
+ Success bool `json:"success"`
+ }
+ if err := json.Unmarshal(result, &response); err != nil || !response.Success {
+ t.Fatalf("pull result = %s, error %v", result, err)
+ }
+ if postBeforeSubscribe.Load() {
+ t.Fatal("download POST arrived before the SSE subscription was open")
+ }
+ var events []ProgressEvent
+ for len(progress) > 0 {
+ events = append(events, <-progress)
+ }
+ if len(events) != 2 {
+ t.Fatalf("progress events = %+v, want downloading and success", events)
+ }
+ if events[0].Stage != "downloading" || events[0].Percent != 50 {
+ t.Fatalf("download event = %+v, want 50%%", events[0])
+ }
+ if events[1].Stage != "success" || events[1].Percent != 100 {
+ t.Fatalf("terminal event = %+v, want success at 100%%", events[1])
+ }
+}
+
+func TestPullModelOllamaAdvancingProgressRefreshesTimeout(t *testing.T) {
+ const (
+ idleTimeout = 400 * time.Millisecond
+ progressDelay = 90 * time.Millisecond
+ progressUpdates = 6
+ )
+ mux := http.NewServeMux()
+ mux.HandleFunc("/api/pull", func(w http.ResponseWriter, _ *http.Request) {
+ flusher, ok := w.(http.Flusher)
+ if !ok {
+ t.Error("test response does not support flushing")
+ return
+ }
+ for completed := 1; completed <= progressUpdates; completed++ {
+ if _, err := fmt.Fprintf(
+ w,
+ "{\"status\":\"pulling\",\"digest\":\"sha256:model\",\"total\":%d,\"completed\":%d}\n",
+ progressUpdates,
+ completed,
+ ); err != nil {
+ return
+ }
+ flusher.Flush()
+ time.Sleep(progressDelay)
+ }
+ _, _ = fmt.Fprintln(w, `{"status":"success"}`)
+ flusher.Flush()
+ })
+ server := httptest.NewServer(mux)
+ defer server.Close()
+
+ ex := newHTTPPullTestExecutor(t, server, Action{
+ HTTP: &ActionHTTP{Method: http.MethodPost, Path: "/api/pull"},
+ })
+ ex.pullProgressTimeout = idleTimeout
+
+ started := time.Now()
+ result, err := ex.PullModelStream(
+ context.Background(),
+ "fake",
+ "demo:1b",
+ json.RawMessage(`{"name":"demo:1b"}`),
+ )
+ if err != nil {
+ t.Fatalf("pull model: %v", err)
+ }
+ if elapsed := time.Since(started); elapsed <= idleTimeout {
+ t.Fatalf("pull completed in %s, want longer than one %s idle interval", elapsed, idleTimeout)
+ }
+ if !strings.Contains(string(result), `"status":"success"`) {
+ t.Fatalf("pull result = %s, want terminal success", result)
+ }
+}
+
+func TestPullModelLlamaCPPAdvancingProgressRefreshesTimeout(t *testing.T) {
+ const (
+ model = "owner/repo:Q4_K_M"
+ idleTimeout = 400 * time.Millisecond
+ progressDelay = 90 * time.Millisecond
+ progressUpdates = 6
+ )
+ started := make(chan struct{})
+ mux := http.NewServeMux()
+ mux.HandleFunc("/models/sse", func(w http.ResponseWriter, r *http.Request) {
+ flusher, ok := w.(http.Flusher)
+ if !ok {
+ t.Error("test response does not support flushing")
+ return
+ }
+ _, _ = fmt.Fprint(w, ": ready\n\n")
+ flusher.Flush()
+ select {
+ case <-started:
+ case <-r.Context().Done():
+ return
+ }
+ for completed := 1; completed <= progressUpdates; completed++ {
+ if _, err := fmt.Fprintf(
+ w,
+ "data: {\"model\":%q,\"event\":\"download_progress\",\"data\":{\"progress\":{\"model.gguf\":{\"done\":%d,\"total\":%d}}}}\n\n",
+ model,
+ completed,
+ progressUpdates,
+ ); err != nil {
+ return
+ }
+ flusher.Flush()
+ time.Sleep(progressDelay)
+ }
+ _, _ = fmt.Fprintf(w, "data: {\"model\":%q,\"event\":\"download_finished\",\"data\":{}}\n\n", model)
+ flusher.Flush()
+ })
+ mux.HandleFunc("/models", func(w http.ResponseWriter, _ *http.Request) {
+ close(started)
+ w.Header().Set("Content-Type", "application/json")
+ _, _ = fmt.Fprint(w, `{"success":true}`)
+ })
+ server := httptest.NewServer(mux)
+ defer server.Close()
+
+ ex := newHTTPPullTestExecutor(t, server, Action{
+ HTTP: &ActionHTTP{Method: http.MethodPost, Path: "/models"},
+ ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE,
+ })
+ ex.pullProgressTimeout = idleTimeout
+
+ startedAt := time.Now()
+ result, err := ex.PullModelStream(
+ context.Background(),
+ "fake",
+ model,
+ json.RawMessage(`{"model":"`+model+`"}`),
+ )
+ if err != nil {
+ t.Fatalf("pull model: %v", err)
+ }
+ if elapsed := time.Since(startedAt); elapsed <= idleTimeout {
+ t.Fatalf("pull completed in %s, want longer than one %s idle interval", elapsed, idleTimeout)
+ }
+ if !strings.Contains(string(result), `"success":true`) {
+ t.Fatalf("pull result = %s, want accepted download response", result)
+ }
+}
+
+func TestPullModelDuplicateProgressDoesNotRefreshTimeout(t *testing.T) {
+ const idleTimeout = 150 * time.Millisecond
+ mux := http.NewServeMux()
+ mux.HandleFunc("/api/pull", func(w http.ResponseWriter, r *http.Request) {
+ flusher, ok := w.(http.Flusher)
+ if !ok {
+ t.Error("test response does not support flushing")
+ return
+ }
+ ticker := time.NewTicker(25 * time.Millisecond)
+ defer ticker.Stop()
+ for {
+ _, err := fmt.Fprintln(w, `{"status":"pulling","digest":"sha256:model","total":100,"completed":1}`)
+ if err != nil {
+ return
+ }
+ flusher.Flush()
+ select {
+ case <-r.Context().Done():
+ return
+ case <-ticker.C:
+ }
+ }
+ })
+ server := httptest.NewServer(mux)
+ defer server.Close()
+
+ ex := newHTTPPullTestExecutor(t, server, Action{
+ HTTP: &ActionHTTP{Method: http.MethodPost, Path: "/api/pull"},
+ })
+ ex.pullProgressTimeout = idleTimeout
+
+ _, err := ex.PullModelStream(
+ context.Background(),
+ "fake",
+ "demo:1b",
+ json.RawMessage(`{"name":"demo:1b"}`),
+ )
+ if err == nil {
+ t.Fatal("pull model succeeded despite duplicate-only progress")
+ }
+ if want := "model download made no progress for 150ms"; !strings.Contains(err.Error(), want) {
+ t.Fatalf("pull error = %q, want %q", err, want)
+ }
+}
+
+func TestPullProgressWatchdogPreservesParentCancellation(t *testing.T) {
+ parent, cancel := context.WithCancel(context.Background())
+ ctx, watchdog := newPullProgressWatchdog(parent, time.Hour)
+ defer watchdog.stop()
+
+ cancel()
+ select {
+ case <-ctx.Done():
+ case <-time.After(time.Second):
+ t.Fatal("watchdog context did not observe parent cancellation")
+ }
+ if cause := context.Cause(ctx); cause != context.Canceled {
+ t.Fatalf("watchdog cause = %v, want context canceled", cause)
+ }
+}
+
// TestActionPullModelStreamsProgress verifies that engine:action with action
// "pull_model" is routed through the streaming pull path, so a local pull
// emits live engine:pull-progress notifications (with computed percentages) and
diff --git a/services/nvpair-engine-manager/registry.go b/services/nvpair-engine-manager/registry.go
index b6a79ad6..c1c1a501 100644
--- a/services/nvpair-engine-manager/registry.go
+++ b/services/nvpair-engine-manager/registry.go
@@ -25,6 +25,11 @@ import (
// schema growth stays backward compatible.
const ManifestSchemaVersion = 1
+const (
+ actionHTTPParamsBody = "body"
+ actionHTTPParamsQuery = "query"
+)
+
// allowedPlaceholders is the set of `{token}`s the runner can resolve
// at execution time. Validation rejects any other token so a typo in
// a manifest fails at load with a clear message rather than at run
@@ -40,6 +45,9 @@ var allowedPlaceholders = map[string]bool{
}
var placeholderRe = regexp.MustCompile(`\{([a-zA-Z_][a-zA-Z0-9_]*)\}`)
+var resultMatchFieldPathRe = regexp.MustCompile(`^[a-zA-Z_][a-zA-Z0-9_-]*(?:\.[a-zA-Z_][a-zA-Z0-9_-]*)*$`)
+var artifactNameRe = regexp.MustCompile(`^[a-z][a-z0-9_]{0,31}$`)
+var sha256Re = regexp.MustCompile(`^[a-fA-F0-9]{64}$`)
// engineNameRe restricts engine names to a safe charset — the name is
// used as a filesystem path component (the per-engine install dir), so
@@ -67,12 +75,13 @@ type Platform struct {
Runtime Runtime `json:"runtime"`
}
-// Install describes how to obtain the engine in user mode. A pinned
-// download (fetch.sha256 set) is checksum-verified before its `run`
-// command executes; an unpinned fetch is HTTPS-only (see download).
+// Install describes how to obtain the engine in user mode. Fetch may be
+// unpinned, but every member of Artifacts is checksum-verified before `run`
+// executes. All downloads are HTTPS-only outside loopback.
type Install struct {
- Fetch *Fetch `json:"fetch,omitempty"`
- Run []string `json:"run,omitempty"`
+ Fetch *Fetch `json:"fetch,omitempty"`
+ Artifacts []InstallArtifact `json:"artifacts,omitempty"`
+ Run []string `json:"run,omitempty"`
// Script is an escape hatch for vendors that only ship a script
// installer. It runs without checksum verification — strictly opt-in
// and logged as unpinned. Prefer fetch+run whenever the vendor publishes
@@ -98,6 +107,14 @@ type Fetch struct {
SHA256 string `json:"sha256"`
}
+// InstallArtifact is one checksum-pinned member of a multi-file install.
+// Its download is available to install.run as {download_}.
+type InstallArtifact struct {
+ Name string `json:"name"`
+ URL string `json:"url"`
+ SHA256 string `json:"sha256"`
+}
+
// Runtime is how to launch + probe the engine. By default the launched
// engine binds loopback; an engine may set Bind (substituted as {host})
// to listen elsewhere — inference engines use "0.0.0.0" to serve the
@@ -164,14 +181,23 @@ func (r *Runtime) hasCustomLaunch() bool {
(r.LaunchEnv != nil && len(*r.LaunchEnv) > 0)
}
-// Probe is an HTTP or TCP reachability check. Exactly one of HTTP/TCP
-// should be set; HTTP wins if both are.
+// ProbeJSONMatch optionally identifies an HTTP service by a string field in
+// its JSON response body. Field supports the same validated dotted object path
+// syntax as action result filters.
+type ProbeJSONMatch struct {
+ Field string `json:"field"`
+ Value string `json:"value"`
+}
+
+// Probe is an HTTP or TCP reachability check. Exactly one of HTTP/TCP should be
+// set; HTTP wins if both are. JSONMatch is valid only for HTTP probes.
type Probe struct {
- HTTP string `json:"http,omitempty"` // url template, e.g. "http://127.0.0.1:{port}/"
- TCP string `json:"tcp,omitempty"` // host:port template, e.g. "127.0.0.1:{port}"
- Status int `json:"status,omitempty"` // expected HTTP status (default 200)
- TimeoutS int `json:"timeout_s,omitempty"`
- IntervalS int `json:"interval_s,omitempty"`
+ HTTP string `json:"http,omitempty"` // url template, e.g. "http://127.0.0.1:{port}/"
+ TCP string `json:"tcp,omitempty"` // host:port template, e.g. "127.0.0.1:{port}"
+ Status int `json:"status,omitempty"` // expected HTTP status (default 200)
+ TimeoutS int `json:"timeout_s,omitempty"`
+ IntervalS int `json:"interval_s,omitempty"`
+ JSONMatch *ProbeJSONMatch `json:"json_match,omitempty"`
}
// StopSpec is how to terminate the engine. Default is a graceful
@@ -186,13 +212,18 @@ type StopSpec struct {
// Exactly one of HTTP (call the engine's loopback control API), Cmd
// (run a CLI command, e.g. `lms get`), or RemovePath (guarded filesystem
// delete) is set. Only a Cmd action templates the action's params as
-// placeholders (e.g. {model}); an HTTP action sends params as the JSON
-// request body.
+// placeholders (e.g. {model}); an HTTP action sends params in its declared
+// body or query location.
type Action struct {
Description string `json:"description,omitempty"`
HTTP *ActionHTTP `json:"http,omitempty"`
Cmd []string `json:"cmd,omitempty"`
RemovePath *ActionRemovePath `json:"remove_path,omitempty"`
+ // ProgressProtocol selects a narrowly defined streaming adapter for an
+ // asynchronous pull HTTP action. Empty keeps the ordinary response-stream
+ // behavior; named protocols are validated so a typo cannot silently fall
+ // back to the wrong completion semantics.
+ ProgressProtocol string `json:"progress_protocol,omitempty"`
// ModelResolution, when set, expands or resolves the model param:
// - "lms-get" on Cmd actions: try as-given → Hub id → Hugging Face URL.
// - "lms-disk-path" on RemovePath actions: map logical ids to on-disk
@@ -228,9 +259,10 @@ type ActionResult struct {
// ResultMatch is the optional row filter on an ActionResult. Exactly one of
// In or Nonempty must be set:
-// - In: keep the element when Field (decoded as a JSON string) equals one of In.
-// - Nonempty: keep the element when Field is a JSON array with length > 0
-// (LM Studio's /api/v1/models models[].loaded_instances).
+// - In: keep the element when the dot-separated object path Field (decoded as
+// a JSON string) equals one of In.
+// - Nonempty: keep the element when Field resolves to a JSON array with length
+// > 0 (LM Studio's /api/v1/models models[].loaded_instances).
type ResultMatch struct {
Field string `json:"field"` // element field to test, e.g. "state" / "loaded_instances"
In []string `json:"in,omitempty"` // accepted string values, e.g. ["loaded"]
@@ -244,12 +276,14 @@ type ActionRemovePath struct {
Root string `json:"root"`
}
-// ActionHTTP is a templated call against the engine's loopback base
-// URL. The caller's params (engine:action params) are sent as the
-// JSON request body; BodySchema is informational only.
+// ActionHTTP is a templated call against the engine's loopback base URL.
+// ParamsIn selects whether the caller's params are sent as the JSON request
+// body (the default) or as URL-encoded query parameters. BodySchema is
+// informational only.
type ActionHTTP struct {
Method string `json:"method"`
Path string `json:"path"`
+ ParamsIn string `json:"params_in,omitempty"`
BodySchema json.RawMessage `json:"body_schema,omitempty"`
}
@@ -636,15 +670,38 @@ func (p *Platform) validate(key string) error {
return fmt.Errorf("platform %q: runtime.mode %q invalid (want \"process\" or \"command\")", key, p.Runtime.Mode)
}
if p.Install != nil {
- if len(p.Install.Script) > 0 && (p.Install.Fetch != nil || len(p.Install.Run) > 0) {
- return fmt.Errorf("platform %q: install.script is mutually exclusive with fetch/run (a script install cannot also be checksum-pinned)", key)
+ hasArtifacts := len(p.Install.Artifacts) > 0
+ if len(p.Install.Script) > 0 && (p.Install.Fetch != nil || hasArtifacts || len(p.Install.Run) > 0) {
+ return fmt.Errorf("platform %q: install.script is mutually exclusive with fetch/artifacts/run (a script install cannot also be checksum-pinned)", key)
+ }
+ if p.Install.Fetch != nil && hasArtifacts {
+ return fmt.Errorf("platform %q: install.fetch and install.artifacts are mutually exclusive", key)
}
- if len(p.Install.Run) > 0 && p.Install.Fetch == nil {
- return fmt.Errorf("platform %q: install.run requires a fetch (the artifact the run command unpacks)", key)
+ if (p.Install.Fetch != nil || hasArtifacts) && len(p.Install.Run) == 0 {
+ return fmt.Errorf("platform %q: install.run is required when install.fetch or install.artifacts is present", key)
+ }
+ if len(p.Install.Run) > 0 && p.Install.Fetch == nil && !hasArtifacts {
+ return fmt.Errorf("platform %q: install.run requires a fetch or artifacts (the downloads the run command uses)", key)
}
if p.Install.Fetch != nil && strings.TrimSpace(p.Install.Fetch.URL) == "" {
return fmt.Errorf("platform %q: install.fetch.url is required when fetch is present", key)
}
+ names := make(map[string]bool, len(p.Install.Artifacts))
+ for index, artifact := range p.Install.Artifacts {
+ if !artifactNameRe.MatchString(artifact.Name) {
+ return fmt.Errorf("platform %q: install.artifacts[%d].name %q must match [a-z][a-z0-9_]{0,31}", key, index, artifact.Name)
+ }
+ if names[artifact.Name] {
+ return fmt.Errorf("platform %q: duplicate install artifact name %q", key, artifact.Name)
+ }
+ names[artifact.Name] = true
+ if err := validateDownloadURL(artifact.URL); err != nil {
+ return fmt.Errorf("platform %q: install artifact %q: %w", key, artifact.Name, err)
+ }
+ if !sha256Re.MatchString(strings.TrimSpace(artifact.SHA256)) {
+ return fmt.Errorf("platform %q: install artifact %q requires a 64-character hexadecimal sha256", key, artifact.Name)
+ }
+ }
switch p.Install.Mode {
case "", "user", "admin":
default:
@@ -674,6 +731,17 @@ func validateProbe(key, which string, p *Probe) error {
if strings.TrimSpace(p.HTTP) == "" && strings.TrimSpace(p.TCP) == "" {
return fmt.Errorf("platform %q: runtime.%s must set either http or tcp", key, which)
}
+ if p.JSONMatch != nil {
+ if strings.TrimSpace(p.HTTP) == "" {
+ return fmt.Errorf("platform %q: runtime.%s json_match requires http", key, which)
+ }
+ if !resultMatchFieldPathRe.MatchString(p.JSONMatch.Field) {
+ return fmt.Errorf("platform %q: runtime.%s json_match.field %q is not a valid object path", key, which, p.JSONMatch.Field)
+ }
+ if strings.TrimSpace(p.JSONMatch.Value) == "" {
+ return fmt.Errorf("platform %q: runtime.%s json_match.value is required", key, which)
+ }
+ }
return nil
}
@@ -702,6 +770,21 @@ func (a *Action) validate(name string) error {
if hasHTTP && (strings.TrimSpace(a.HTTP.Method) == "" || strings.TrimSpace(a.HTTP.Path) == "") {
return fmt.Errorf("action %q: http.method and http.path are required", name)
}
+ if hasHTTP {
+ switch a.HTTP.ParamsIn {
+ case "", actionHTTPParamsBody, actionHTTPParamsQuery:
+ default:
+ return fmt.Errorf("action %q: http.params_in %q invalid (want %q or %q)", name, a.HTTP.ParamsIn, actionHTTPParamsBody, actionHTTPParamsQuery)
+ }
+ }
+ if a.ProgressProtocol != "" {
+ if name != pullModelAction || !hasHTTP {
+ return fmt.Errorf("action %q: progress_protocol requires the HTTP pull_model action", name)
+ }
+ if a.ProgressProtocol != pullProgressProtocolLlamaCPPModelsSSE {
+ return fmt.Errorf("action %q: unsupported progress_protocol %q", name, a.ProgressProtocol)
+ }
+ }
if a.Result != nil && (strings.TrimSpace(a.Result.Array) == "" || strings.TrimSpace(a.Result.Field) == "") {
return fmt.Errorf("action %q: result.array and result.field are required when result is set", name)
}
@@ -710,6 +793,9 @@ func (a *Action) validate(name string) error {
if strings.TrimSpace(m.Field) == "" {
return fmt.Errorf("action %q: result.match.field is required when result.match is set", name)
}
+ if !resultMatchFieldPathRe.MatchString(m.Field) {
+ return fmt.Errorf("action %q: result.match.field %q is not a valid object path", name, m.Field)
+ }
hasIn := len(m.In) > 0
if hasIn == m.Nonempty {
return fmt.Errorf("action %q: result.match requires exactly one of a non-empty in or nonempty=true", name)
@@ -735,57 +821,76 @@ func (a *Action) validate(name string) error {
return nil
}
-// validatePlaceholders rejects any `{token}` outside allowedPlaceholders
-// across every templated string in the manifest.
+// validatePlaceholders rejects any `{token}` outside the placeholders
+// available to the platform or manifest-global action that contains it.
func (m *Manifest) validatePlaceholders() error {
- for _, s := range m.templatedStrings() {
- for _, match := range placeholderRe.FindAllStringSubmatch(s, -1) {
- if !allowedPlaceholders[match[1]] {
- return fmt.Errorf("unknown placeholder {%s} (allowed: %s)", match[1], strings.Join(allowedPlaceholderList(), ", "))
+ for key, platform := range m.Platforms {
+ allowed := make(map[string]bool, len(allowedPlaceholders))
+ for name := range allowedPlaceholders {
+ allowed[name] = true
+ }
+ if platform.Install != nil {
+ for _, artifact := range platform.Install.Artifacts {
+ allowed["download_"+artifact.Name] = true
+ }
+ }
+ if err := validatePlaceholderStrings(platform.templatedStrings(), allowed); err != nil {
+ return fmt.Errorf("platform %q: %w", key, err)
+ }
+ }
+ return validatePlaceholderStrings(m.actionTemplatedStrings(), allowedPlaceholders)
+}
+
+func validatePlaceholderStrings(templates []string, allowed map[string]bool) error {
+ for _, template := range templates {
+ for _, match := range placeholderRe.FindAllStringSubmatch(template, -1) {
+ if !allowed[match[1]] {
+ return fmt.Errorf("unknown placeholder {%s} (allowed: %s)", match[1], strings.Join(placeholderList(allowed), ", "))
}
}
}
return nil
}
-// allowedPlaceholderList returns the allowed placeholder names, sorted,
-// so error messages can't drift from the actual allow-set.
-func allowedPlaceholderList() []string {
- out := make([]string, 0, len(allowedPlaceholders))
- for k := range allowedPlaceholders {
+func placeholderList(placeholders map[string]bool) []string {
+ out := make([]string, 0, len(placeholders))
+ for k := range placeholders {
out = append(out, k)
}
sort.Strings(out)
return out
}
-// templatedStrings collects every string the runner resolves
-// placeholders in, so validatePlaceholders can scan them all.
-func (m *Manifest) templatedStrings() []string {
+// templatedStrings collects every platform-local string the runner resolves
+// placeholders in, so artifact placeholders stay scoped to their platform.
+func (p Platform) templatedStrings() []string {
var out []string
- for _, p := range m.Platforms {
- out = append(out, p.Detect...)
- if p.Install != nil {
- out = append(out, p.Install.Run...)
- out = append(out, p.Install.Script...)
- }
- if p.Uninstall != nil {
- out = append(out, p.Uninstall.Run...)
- }
- out = append(out, p.Runtime.Bin)
- out = append(out, p.Runtime.Args...)
- for _, cmd := range p.Runtime.Start {
- out = append(out, cmd...)
- }
- for _, v := range p.Runtime.Env {
- out = append(out, v)
- }
- out = append(out, probeStrings(p.Runtime.Ready)...)
- out = append(out, probeStrings(p.Runtime.Health)...)
- if p.Runtime.Stop != nil {
- out = append(out, p.Runtime.Stop.Cmd...)
- }
+ out = append(out, p.Detect...)
+ if p.Install != nil {
+ out = append(out, p.Install.Run...)
+ out = append(out, p.Install.Script...)
+ }
+ if p.Uninstall != nil {
+ out = append(out, p.Uninstall.Run...)
+ }
+ out = append(out, p.Runtime.Bin)
+ out = append(out, p.Runtime.Args...)
+ for _, cmd := range p.Runtime.Start {
+ out = append(out, cmd...)
}
+ for _, v := range p.Runtime.Env {
+ out = append(out, v)
+ }
+ out = append(out, probeStrings(p.Runtime.Ready)...)
+ out = append(out, probeStrings(p.Runtime.Health)...)
+ if p.Runtime.Stop != nil {
+ out = append(out, p.Runtime.Stop.Cmd...)
+ }
+ return out
+}
+
+func (m *Manifest) actionTemplatedStrings() []string {
+ var out []string
for _, act := range m.Actions {
if act.RemovePath != nil {
out = append(out, act.RemovePath.Root)
diff --git a/services/nvpair-engine-manager/registry_test.go b/services/nvpair-engine-manager/registry_test.go
index 2ae2c5c6..26859d85 100644
--- a/services/nvpair-engine-manager/registry_test.go
+++ b/services/nvpair-engine-manager/registry_test.go
@@ -5,6 +5,7 @@ package main
import (
"encoding/json"
+ "net/http"
"os"
"path/filepath"
"slices"
@@ -69,6 +70,74 @@ func TestValidateAcceptsCommandModeAndCmdAction(t *testing.T) {
}
}
+func TestValidateAcceptsLlamaCPPModelPullProtocol(t *testing.T) {
+ m := validManifest()
+ m.Actions[pullModelAction] = Action{
+ HTTP: &ActionHTTP{Method: "POST", Path: "/models"},
+ ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE,
+ }
+ if err := m.Validate(); err != nil {
+ t.Fatalf("llama.cpp model pull protocol rejected: %v", err)
+ }
+}
+
+func TestValidateAcceptsHTTPQueryParams(t *testing.T) {
+ m := validManifest()
+ m.Actions["delete_model"] = Action{
+ HTTP: &ActionHTTP{
+ Method: http.MethodDelete,
+ Path: "/models",
+ ParamsIn: actionHTTPParamsQuery,
+ },
+ }
+ if err := m.Validate(); err != nil {
+ t.Fatalf("HTTP query params rejected: %v", err)
+ }
+}
+
+func TestValidateRejectsInvalidHTTPParamsLocation(t *testing.T) {
+ m := validManifest()
+ m.Actions["delete_model"] = Action{
+ HTTP: &ActionHTTP{
+ Method: http.MethodDelete,
+ Path: "/models",
+ ParamsIn: "headers",
+ },
+ }
+ err := m.Validate()
+ if err == nil {
+ t.Fatal("invalid HTTP params location accepted")
+ }
+ if !strings.Contains(err.Error(), "http.params_in") {
+ t.Fatalf("error = %q, want http.params_in", err)
+ }
+}
+
+func TestValidateAcceptsNestedResultMatch(t *testing.T) {
+ m := validManifest()
+ m.Actions["list_models"] = Action{
+ HTTP: &ActionHTTP{Method: "GET", Path: "/models"},
+ Result: &ActionResult{
+ Array: "data",
+ Field: "id",
+ Match: &ResultMatch{Field: "status.value", In: []string{"loaded"}},
+ },
+ }
+ if err := m.Validate(); err != nil {
+ t.Fatalf("nested result match rejected: %v", err)
+ }
+}
+
+func TestValidateAcceptsProbeJSONMatch(t *testing.T) {
+ m := validManifest()
+ p := m.Platforms["linux/amd64"]
+ p.Runtime.Ready.JSONMatch = &ProbeJSONMatch{Field: "service.role", Value: "router"}
+ m.Platforms["linux/amd64"] = p
+ if err := m.Validate(); err != nil {
+ t.Fatalf("HTTP probe JSON match rejected: %v", err)
+ }
+}
+
func TestValidateAcceptsUnpinnedFetch(t *testing.T) {
m := validManifest()
p := m.Platforms["linux/amd64"]
@@ -79,6 +148,101 @@ func TestValidateAcceptsUnpinnedFetch(t *testing.T) {
}
}
+func TestValidateAcceptsNamedInstallArtifacts(t *testing.T) {
+ m := validManifest()
+ setInstallArtifacts(&m, validInstallArtifacts())
+ if err := m.Validate(); err != nil {
+ t.Fatalf("named install artifacts rejected: %v", err)
+ }
+}
+
+func TestValidateRejectsDownloadInstallWithoutRun(t *testing.T) {
+ test := func(name string, install *Install) {
+ t.Run(name, func(t *testing.T) {
+ m := validManifest()
+ p := m.Platforms["linux/amd64"]
+ p.Install = install
+ m.Platforms["linux/amd64"] = p
+
+ err := m.Validate()
+ if err == nil {
+ t.Fatal("download install without run accepted")
+ }
+ const want = `platform "linux/amd64": install.run is required when install.fetch or install.artifacts is present`
+ if err.Error() != want {
+ t.Fatalf("validation error = %q, want %q", err, want)
+ }
+ })
+ }
+
+ test("fetch with omitted run", &Install{
+ Fetch: &Fetch{URL: "https://example/installer.zip"},
+ })
+ test("fetch with empty run", &Install{
+ Fetch: &Fetch{URL: "https://example/installer.zip"},
+ Run: []string{},
+ })
+ test("artifacts with omitted run", &Install{
+ Artifacts: validInstallArtifacts(),
+ })
+ test("artifacts with empty run", &Install{
+ Artifacts: validInstallArtifacts(),
+ Run: []string{},
+ })
+}
+
+func TestValidateAcceptsScriptOnlyInstall(t *testing.T) {
+ m := validManifest()
+ p := m.Platforms["linux/amd64"]
+ p.Install = &Install{Script: []string{"sh", "installer.sh"}}
+ m.Platforms["linux/amd64"] = p
+ if err := m.Validate(); err != nil {
+ t.Fatalf("script-only install rejected: %v", err)
+ }
+}
+
+func TestValidateRejectsArtifactPlaceholderFromAnotherPlatform(t *testing.T) {
+ m := validManifest()
+ setInstallArtifacts(&m, validInstallArtifacts())
+ mac := Platform{
+ Install: &Install{
+ Fetch: &Fetch{URL: "https://example/server.tar.gz"},
+ Run: []string{"extract", "{download}"},
+ },
+ Runtime: Runtime{Bin: "{install_dir}/llama-server"},
+ }
+ m.Platforms["darwin/arm64"] = mac
+ if err := m.Validate(); err != nil {
+ t.Fatalf("valid multi-platform fixture rejected: %v", err)
+ }
+
+ mac.Install.Run = []string{"extract", "{download_cudart}"}
+ m.Platforms["darwin/arm64"] = mac
+ err := m.Validate()
+ if err == nil {
+ t.Fatal("macOS install references {download_cudart}, but only Linux declares cudart; want validation error")
+ }
+ const want = `platform "darwin/arm64": unknown placeholder {download_cudart}`
+ if !strings.Contains(err.Error(), want) {
+ t.Fatalf("error = %q, want it to contain %q", err, want)
+ }
+}
+
+func validInstallArtifacts() []InstallArtifact {
+ return []InstallArtifact{
+ {Name: "server", URL: "https://example/server.zip", SHA256: strings.Repeat("a", 64)},
+ {Name: "cudart", URL: "https://example/cudart.zip", SHA256: strings.Repeat("b", 64)},
+ }
+}
+
+func setInstallArtifacts(m *Manifest, artifacts []InstallArtifact) {
+ p := m.Platforms["linux/amd64"]
+ p.Install.Fetch = nil
+ p.Install.Artifacts = artifacts
+ p.Install.Run = []string{"extract", "{download_server}", "{download_cudart}"}
+ m.Platforms["linux/amd64"] = p
+}
+
func TestValidateRejectsBadEngineName(t *testing.T) {
for _, bad := range []string{"../evil", "a/b", `a\b`, "..", ".", "a b", ""} {
m := validManifest()
@@ -123,6 +287,37 @@ func TestValidateRejects(t *testing.T) {
p.Install.Fetch = nil
m.Platforms["linux/amd64"] = p
}, "requires a fetch"},
+ {"fetch with artifacts", func(m *Manifest) {
+ p := m.Platforms["linux/amd64"]
+ p.Install.Artifacts = validInstallArtifacts()
+ m.Platforms["linux/amd64"] = p
+ }, "mutually exclusive"},
+ {"invalid artifact name", func(m *Manifest) {
+ artifacts := validInstallArtifacts()
+ artifacts[0].Name = "../server"
+ setInstallArtifacts(m, artifacts)
+ }, "must match"},
+ {"duplicate artifact name", func(m *Manifest) {
+ artifacts := validInstallArtifacts()
+ artifacts[1].Name = artifacts[0].Name
+ setInstallArtifacts(m, artifacts)
+ }, "duplicate install artifact"},
+ {"insecure artifact URL", func(m *Manifest) {
+ artifacts := validInstallArtifacts()
+ artifacts[0].URL = "http://example.com/server.zip"
+ setInstallArtifacts(m, artifacts)
+ }, "must be https"},
+ {"invalid artifact checksum", func(m *Manifest) {
+ artifacts := validInstallArtifacts()
+ artifacts[0].SHA256 = "deadbeef"
+ setInstallArtifacts(m, artifacts)
+ }, "64-character hexadecimal"},
+ {"unknown artifact placeholder", func(m *Manifest) {
+ setInstallArtifacts(m, validInstallArtifacts())
+ p := m.Platforms["linux/amd64"]
+ p.Install.Run = append(p.Install.Run, "{download_gpu}")
+ m.Platforms["linux/amd64"] = p
+ }, "unknown placeholder {download_gpu}"},
{"bad install mode", func(m *Manifest) {
p := m.Platforms["linux/amd64"]
p.Install.Mode = "root"
@@ -133,12 +328,52 @@ func TestValidateRejects(t *testing.T) {
p.Runtime.Args = []string{"serve", "{bogus}"}
m.Platforms["linux/amd64"] = p
}, "unknown placeholder {bogus}"},
+ {"JSON match without HTTP", func(m *Manifest) {
+ p := m.Platforms["linux/amd64"]
+ p.Runtime.Ready = &Probe{
+ TCP: "127.0.0.1:{port}",
+ JSONMatch: &ProbeJSONMatch{Field: "role", Value: "router"},
+ }
+ m.Platforms["linux/amd64"] = p
+ }, "json_match requires http"},
+ {"invalid JSON match field path", func(m *Manifest) {
+ p := m.Platforms["linux/amd64"]
+ p.Runtime.Ready.JSONMatch = &ProbeJSONMatch{Field: "service..role", Value: "router"}
+ m.Platforms["linux/amd64"] = p
+ }, "is not a valid object path"},
+ {"missing JSON match value", func(m *Manifest) {
+ p := m.Platforms["linux/amd64"]
+ p.Runtime.Ready.JSONMatch = &ProbeJSONMatch{Field: "role"}
+ m.Platforms["linux/amd64"] = p
+ }, "json_match.value is required"},
{"action without http or cmd", func(m *Manifest) {
m.Actions = map[string]Action{"x": {Description: "neither"}}
}, "exactly one of http, cmd, or remove_path"},
{"action missing method", func(m *Manifest) {
m.Actions = map[string]Action{"x": {HTTP: &ActionHTTP{Path: "/p"}}}
}, "http.method and http.path"},
+ {"unknown progress protocol", func(m *Manifest) {
+ m.Actions[pullModelAction] = Action{
+ HTTP: &ActionHTTP{Method: "POST", Path: "/models"},
+ ProgressProtocol: "unknown",
+ }
+ }, "unsupported progress_protocol"},
+ {"progress protocol on other action", func(m *Manifest) {
+ m.Actions["list_models"] = Action{
+ HTTP: &ActionHTTP{Method: "GET", Path: "/models"},
+ ProgressProtocol: pullProgressProtocolLlamaCPPModelsSSE,
+ }
+ }, "requires the HTTP pull_model action"},
+ {"invalid match field path", func(m *Manifest) {
+ m.Actions["list_models"] = Action{
+ HTTP: &ActionHTTP{Method: "GET", Path: "/models"},
+ Result: &ActionResult{
+ Array: "models",
+ Field: "id",
+ Match: &ResultMatch{Field: "status..value", In: []string{"loaded"}},
+ },
+ }
+ }, "is not a valid object path"},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
@@ -227,6 +462,40 @@ func TestLoadRegistryOverride(t *testing.T) {
}
}
+func TestLoadOverrideDirRejectsEmptyInstallRun(t *testing.T) {
+ reg := NewRegistry()
+ if err := reg.LoadFS(bundledManifests, "manifests"); err != nil {
+ t.Fatalf("load bundled manifests: %v", err)
+ }
+ base, ok := reg.Get("ollama")
+ if !ok {
+ t.Fatal("bundled ollama manifest missing")
+ }
+ baseRun := base.Platforms["linux/amd64"].Install.Run
+ if len(baseRun) == 0 {
+ t.Fatal("bundled ollama install.run is empty")
+ }
+ dir := t.TempDir()
+ const override = `{
+ "engine": "ollama",
+ "display_name": "Invalid override",
+ "platforms": {"linux/amd64": {"install": {"run": []}}}
+}`
+ if err := os.WriteFile(filepath.Join(dir, "ollama.json"), []byte(override), 0o644); err != nil {
+ t.Fatalf("write override: %v", err)
+ }
+ if err := reg.LoadOverrideDir(dir); err != nil {
+ t.Fatalf("load override directory: %v", err)
+ }
+ got, ok := reg.Get("ollama")
+ if !ok || got != base {
+ t.Fatalf("invalid override replaced bundled manifest: got %+v, want original manifest", got)
+ }
+ if !slices.Equal(got.Platforms["linux/amd64"].Install.Run, baseRun) {
+ t.Fatal("invalid override changed the bundled install.run")
+ }
+}
+
func TestLoadRegistryRejectsInvalidFile(t *testing.T) {
dir := t.TempDir()
if err := os.WriteFile(filepath.Join(dir, "broken.json"), []byte(`{"engine":"x"}`), 0o644); err != nil {
@@ -468,6 +737,57 @@ func TestLMStudioManifestBindsLoopback(t *testing.T) {
}
}
+func TestLlamaCPPManifestRequiresRouterIdentity(t *testing.T) {
+ reg := NewRegistry()
+ if err := reg.LoadFS(bundledManifests, "manifests"); err != nil {
+ t.Fatal(err)
+ }
+ m, ok := reg.Get("llamacpp")
+ if !ok {
+ t.Fatal("llamacpp manifest not loaded")
+ }
+ for key, p := range m.Platforms {
+ ready := p.Runtime.Ready
+ if ready == nil || ready.JSONMatch == nil {
+ t.Errorf("%s: readiness JSON identity is missing", key)
+ continue
+ }
+ if ready.HTTP != "http://127.0.0.1:{port}/props" {
+ t.Errorf("%s: readiness URL = %q, want router /props", key, ready.HTTP)
+ }
+ if ready.JSONMatch.Field != "role" || ready.JSONMatch.Value != "router" {
+ t.Errorf("%s: readiness JSON identity = %+v, want role=router", key, ready.JSONMatch)
+ }
+ if health := p.Runtime.Health; health == nil || health.HTTP != "http://127.0.0.1:{port}/health" || health.JSONMatch != nil {
+ t.Errorf("%s: ongoing health probe changed unexpectedly: %+v", key, health)
+ }
+ }
+}
+
+func TestLlamaCPPManifestDeclaresNativeCacheDelete(t *testing.T) {
+ reg := NewRegistry()
+ if err := reg.LoadFS(bundledManifests, "manifests"); err != nil {
+ t.Fatal(err)
+ }
+ m, ok := reg.Get("llamacpp")
+ if !ok {
+ t.Fatal("llamacpp manifest not loaded")
+ }
+ action, ok := m.Actions["delete_model"]
+ if !ok || action.HTTP == nil {
+ t.Fatalf("llamacpp delete_model action is incomplete: %+v", action)
+ }
+ if action.HTTP.Method != http.MethodDelete || action.HTTP.Path != "/models" {
+ t.Errorf("llamacpp delete_model HTTP = %s %s, want DELETE /models", action.HTTP.Method, action.HTTP.Path)
+ }
+ if action.HTTP.ParamsIn != actionHTTPParamsQuery {
+ t.Errorf("llamacpp delete_model params_in = %q, want %q", action.HTTP.ParamsIn, actionHTTPParamsQuery)
+ }
+ if action.RestartAfter {
+ t.Error("llamacpp delete_model must not restart the router")
+ }
+}
+
func TestLMStudioManifestUsesNativeSystemInventory(t *testing.T) {
reg := NewRegistry()
if err := reg.LoadFS(bundledManifests, "manifests"); err != nil {
diff --git a/services/nvpair-engine-manager/remediation_test.go b/services/nvpair-engine-manager/remediation_test.go
index 1999373b..b47478b7 100644
--- a/services/nvpair-engine-manager/remediation_test.go
+++ b/services/nvpair-engine-manager/remediation_test.go
@@ -404,6 +404,7 @@ func TestExpandPathForms(t *testing.T) {
{"unix dollar", "linux", "a/$NVPAIR_TEST_VAR/b", "a/xyz/b"},
{"unix braced dollar", "linux", "a/${NVPAIR_TEST_VAR}/b", "a/xyz/b"},
{"unix percent", "linux", "a/%NVPAIR_TEST_VAR%/b", "a/%NVPAIR_TEST_VAR%/b"},
+ {"unix shell parameters", "linux", `tar -xzf "$1" -C "$3"`, `tar -xzf "$1" -C "$3"`},
} {
t.Run(tc.name, func(t *testing.T) {
if got := expandPathForOS(tc.input, tc.goos); got != tc.want {
@@ -453,7 +454,7 @@ func TestBundledManifestsGolden(t *testing.T) {
if err := reg.LoadFS(bundledManifests, "manifests"); err != nil {
t.Fatalf("bundled manifests invalid: %v", err)
}
- for _, want := range []string{"ollama", "lmstudio"} {
+ for _, want := range []string{"ollama", "lmstudio", "llamacpp"} {
m, ok := reg.Get(want)
if !ok {
t.Fatalf("missing bundled engine %q (have %v)", want, reg.Names())
diff --git a/services/nvpair-engine-manager/settingsremote.go b/services/nvpair-engine-manager/settingsremote.go
index 07689628..53865db7 100644
--- a/services/nvpair-engine-manager/settingsremote.go
+++ b/services/nvpair-engine-manager/settingsremote.go
@@ -13,6 +13,7 @@ import (
"strings"
"time"
+ "nvpair-shared/engines"
settings "nvpair-shared/enginesettings"
)
@@ -204,6 +205,16 @@ func (m *Manager) watchPeerSettings(ctx context.Context) {
}
}
+func (m *Manager) notifyPeerSettings(nodeID string, snapshots []settings.Snapshot) {
+ for _, snapshot := range snapshots {
+ if _, ok := engines.ByName(snapshot.Engine); !ok {
+ continue
+ }
+ snapshot.NodeID = nodeID
+ m.exec.notify("engine:settings-changed", snapshot)
+ }
+}
+
func (m *Manager) consumeSettingsStream(parent context.Context, peer ecPeer) {
ctx, cancel := context.WithCancel(parent)
defer cancel()
@@ -234,12 +245,6 @@ func (m *Manager) consumeSettingsStream(parent context.Context, peer ecPeer) {
if json.Unmarshal(scanner.Bytes(), &snapshots) != nil {
return
}
- for _, snapshot := range snapshots {
- if snapshot.Engine != "ollama" && snapshot.Engine != "lmstudio" {
- continue
- }
- snapshot.NodeID = peer.nodeID
- m.exec.notify("engine:settings-changed", snapshot)
- }
+ m.notifyPeerSettings(peer.nodeID, snapshots)
}
}
diff --git a/services/nvpair-engine-manager/settingsremote_test.go b/services/nvpair-engine-manager/settingsremote_test.go
index 0cdca747..d13dddad 100644
--- a/services/nvpair-engine-manager/settingsremote_test.go
+++ b/services/nvpair-engine-manager/settingsremote_test.go
@@ -64,6 +64,44 @@ func settingsPin(t *testing.T, dir, id string, cert []byte) {
}
}
+func TestNotifyPeerSettingsForwardsEverySupportedEngine(t *testing.T) {
+ emitted := []settings.Snapshot{}
+ exec := &Executor{emit: func(method string, value any) {
+ if method != "engine:settings-changed" {
+ t.Fatalf("notification method = %q, want engine:settings-changed", method)
+ }
+ data, err := json.Marshal(value)
+ if err != nil {
+ t.Fatalf("marshal notification: %v", err)
+ }
+ var snapshot settings.Snapshot
+ if err := json.Unmarshal(data, &snapshot); err != nil {
+ t.Fatalf("decode notification: %v", err)
+ }
+ emitted = append(emitted, snapshot)
+ }}
+ manager := &Manager{exec: exec}
+ manager.notifyPeerSettings("peer-node", []settings.Snapshot{
+ {NodeID: "untrusted-node", Engine: "ollama"},
+ {NodeID: "untrusted-node", Engine: "lmstudio"},
+ {NodeID: "untrusted-node", Engine: "llamacpp"},
+ {NodeID: "untrusted-node", Engine: "unsupported"},
+ })
+
+ wantEngines := []string{"ollama", "lmstudio", "llamacpp"}
+ if len(emitted) != len(wantEngines) {
+ t.Fatalf("emitted %d snapshots, want %d: %+v", len(emitted), len(wantEngines), emitted)
+ }
+ for i, wantEngine := range wantEngines {
+ if emitted[i].Engine != wantEngine {
+ t.Errorf("snapshot %d engine = %q, want %q", i, emitted[i].Engine, wantEngine)
+ }
+ if emitted[i].NodeID != "peer-node" {
+ t.Errorf("snapshot %d nodeId = %q, want peer-node", i, emitted[i].NodeID)
+ }
+ }
+}
+
func TestSettingsPairedMutationPushObserversAndRevocation(t *testing.T) {
a, certA := settingsMesh(t, "a")
b, certB := settingsMesh(t, "b")
diff --git a/services/nvpair-engine-manager/spec.md b/services/nvpair-engine-manager/spec.md
index ba2e6324..cdaa178b 100644
--- a/services/nvpair-engine-manager/spec.md
+++ b/services/nvpair-engine-manager/spec.md
@@ -6,7 +6,13 @@ SPDX-License-Identifier: Apache-2.0
# Microservice: Engine Manager (`nvpair-engine-manager`)
## 1. Purpose
-A declarative, config-driven control plane for **local inference engines** (Ollama today; Intel / llama.cpp / others later). It owns an engine's entire lifecycle *except serving inference* — locate, install, launch, stop, restart, health, and config-declared actions — so one uniform API manages any engine across OSes with no per-engine code. A third party drops in a JSON manifest and their engine's install/launch/controls "just appear" over the same API: the core extensibility story for an open-source product.
+A declarative, config-driven control plane for **local inference engines**
+(currently Ollama, LM Studio, and llama.cpp). It owns an engine's entire
+lifecycle *except serving inference* — locate, install, launch, stop, restart,
+health, and config-declared actions — so one uniform API manages any engine
+across OSes with no per-engine code. A third party drops in a JSON manifest and
+their engine's install/launch/controls "just appear" over the same API: the core
+extensibility story for an open-source product.
## 2. Scope
**In scope**
@@ -22,7 +28,26 @@ A declarative, config-driven control plane for **local inference engines** (Olla
host-platform port overrides, deleting the file only when it has no other
settings. Host-platform precedence must not override a successfully saved port
on reload; malformed existing overrides fail the save and remain intact.
-- Config-declared **actions** covering the full model lifecycle — Ollama: `list_models`, `loaded_models`, `pull_model`, `run_model`, `unload_model`, `delete_model`; LM Studio: `list_models`/`list_downloaded`, `loaded_models`, `pull_model`, `load_model`, `chat`, `unload_model`, `delete_model` (`remove_path` with `lms-disk-path` resolution) — mapped to each engine's local control API. `loaded_models` reports the models currently resident in memory (Ollama `GET /api/ps`, LM Studio `GET /api/v1/models` filtered by nonempty `loaded_instances`), name-extracted via the same declarative `result` spec (with an optional `match` row filter).
+- Config-declared **actions** mapped to each engine's local control API:
+ Ollama declares `list_models`, `loaded_models`, `pull_model`, `run_model`,
+ `unload_model`, and `delete_model`; LM Studio declares
+ `list_models`/`list_downloaded`, `loaded_models`, `pull_model`, `load_model`,
+ `chat`, `unload_model`, and `delete_model` (`remove_path` with
+ `lms-disk-path` resolution);
+ llama.cpp declares list, loaded-list, pull, load, unload, and native cache
+ delete with exact model ids. `loaded_models` reports
+ models currently resident in memory, name-extracted via the same declarative
+ `result` spec with optional nested-path and row filters: Ollama uses
+ `GET /api/ps`, LM Studio filters nonempty `loaded_instances` from
+ `GET /api/v1/models`, and llama.cpp matches nested `status.value == loaded`
+ records from `GET /models`.
+- The bundled llama.cpp launch fixes `--sleep-idle-seconds 300`: after five
+ minutes without inference work, a loaded model releases its model and
+ KV-cache memory and wakes on the next request. The router child remains alive
+ and can retain a residual backend GPU context. Its readiness probe requires
+ `/props` to report `role:"router"`; a single-model listener is incompatible
+ and must not be started or adopted because its `/models` contract does not
+ expose router residency state.
- Per-engine stdout/stderr log capture and structured operational error records, surfaced via the errors pipeline.
- A normalized node-level model list (`engine:models`): union of every running engine's `list_models`, name-extracted via each action's declarative `result` spec, plus the per-engine set of models loaded in memory (`loadedByEngine`, from each engine's `loaded_models` action). A successful explicit empty inventory remains an engine key with `[]`; a missing/malformed/failed inventory omits that engine key instead of being mislabeled as authoritative empty. A watcher polls the loaded set and pushes `engine:models-changed` when it changes (explicit load/unload, JIT auto-load, TTL/idle eviction).
- Expose all of the above over the `engine:*` JSON-RPC surface to whatever orchestrates the service, plus an optional plain-HTTP LAN endpoint (`--http-port`, `GET /v1/models`) that serves the model list to a peer's discovery daemon (the list moved off the size-limited mDNS TXT onto HTTP).
@@ -31,6 +56,7 @@ A declarative, config-driven control plane for **local inference engines** (Olla
- **Inference traffic** — stays with `nvpair-proxy`; this service never proxies `/api/chat` etc.
- **Multi-instance per engine and an MCP server** — future-additive, not v1.
- **The node's error list** — owned by `nvpair-errors`, which holds it as in-memory session state; this service only emits `errors:report` / `errors:clear`.
+- Automatic deletion of persistent llama.cpp model-cache files.
## 3. Key Use Cases
- **Install an engine, user-mode**: `engine:install {engine:"ollama"}` downloads the per-OS user-scoped package (Windows/Linux standalone archive extracted into a user dir; macOS app bundle — never an elevated `Setup.exe` or `curl | sh`), checksum-verifies, extracts, re-detects.
@@ -49,13 +75,14 @@ A declarative, config-driven control plane for **local inference engines** (Olla
- **Risk — admin-only installers**: `mode: "admin"` is a flagged, refused exception, not the default; product direction is strictly user-mode.
- **Risk — LAN-open inference bind (interim)**: inference engines default `runtime.bind` to `0.0.0.0` (ordinary engines stay loopback; a per-call `bind` re-pins). This is a deliberate, temporary exception to the loopback-only posture; narrow it once authenticated inference transport exists. The engine-manager `ec` control surface is already protected independently by pin-based mTLS.
- **Future — declared/tunable env layer**: `runtime.env` is static today. A "declared tunables" layer (manifest-declared knobs, UI/broker-overridable per start) is worth adding; by env-richness the priority is Ollama → llama.cpp/Jan → vLLM (LM Studio / GPT4All are flags/settings-driven, not env). Related: `runtime.env` is process-mode-only — extending it to command-mode start commands is a deliberate, still-open choice. See `MANIFEST.md` → "Engine config reference".
-- **Model deletion where the vendor has no command (LM Studio)**: implemented via the generic **`remove_path`** action kind — a manifest-declared, param-templated path the runner removes with safety rails (must resolve under a declared allowed root, reject `..`/symlink escapes). LM Studio's `delete_model` uses `model_resolution: "lms-disk-path"` to map logical ids to on-disk files via `lms ls --json` before deleting under `{models_dir}`, then `restart_after` to bounce a running server: LM Studio answers `/v1/models` from an index built at startup and exposes no rescan, so clients keep being offered the deleted model until it restarts. Ollama deletes via `DELETE /api/delete` (no restart needed) and ejects via `unload_model` (`POST /api/generate` with `keep_alive: 0`).
+- **Model deletion where the vendor has no command (LM Studio)**: implemented via the generic **`remove_path`** action kind — a manifest-declared, param-templated path the runner removes with safety rails (must resolve under a declared allowed root, reject `..`/symlink escapes). LM Studio's `delete_model` uses `model_resolution: "lms-disk-path"` to map logical ids to on-disk files via `lms ls --json` before deleting under `{models_dir}`, then `restart_after` to bounce a running server: LM Studio answers `/v1/models` from an index built at startup and exposes no rescan, so clients keep being offered the deleted model until it restarts. Ollama deletes via `DELETE /api/delete` (no restart needed) and ejects via `unload_model` (`POST /api/generate` with `keep_alive: 0`). llama.cpp deletes native cache entries via `DELETE /models` with the exact model id in a URL-encoded query parameter; the router updates its inventory in-process, so it also needs no restart.
## 5. Requirements
**Functional**
-- Load + validate per-engine JSON manifests (bundled + user dir); select the host `/` block; resolve placeholders (`{bin}`, `{cli}`, `{port}`, `{download}`, `{install_dir}`).
+- Load + validate per-engine JSON manifests (bundled + user dir); select the host `/` block; resolve placeholders (`{bin}`, `{cli}`, `{port}`, `{download}`, `{install_dir}`). Install commands also receive resolved download and destination paths in child-scoped `NVPAIR_INSTALL_*` environment variables so shell reparsing cannot corrupt them.
- Support both `process` (owned foreground) and `command` (daemon + control-CLI) runtimes; execute detect / install / uninstall / start / stop / restart / status / health and HTTP **or** CLI actions; emit `engine:*` results and notifications.
+- Bound process-mode stops: on Unix send SIGTERM to the owned process group, then SIGKILL after `runtime.stop.grace_s` (five seconds by default); `signal:"kill"` skips the grace. On Windows, windowless managed engines require immediate `taskkill /T /F`.
- Emit `errors:report` / `errors:clear` on its stdio for the Broker to forward to `nvpair-errors`.
**Non-functional**
@@ -160,7 +187,12 @@ The `engine:remote-*` methods are the client half: engine-manager resolves the t
## 9. Data Ownership
- **Owned**: the in-memory engine registry (parsed manifests + per-engine runtime state) and per-engine log/error ring buffers — transient only.
- **Source of truth**: no — `nvpair-errors` owns the node's error list (in memory, for the session); model inventories belong to the engines; manifests on disk are authored elsewhere.
-- **Storage**: in-memory; manifests read from the per-user data dir's `engines/*.json` (`%LocalAppData%\Nvidia Corporation\Personal AI Router` on Windows, `~/.config/Nvidia Corporation/Personal AI Router` on Linux, `~/Library/Application Support/Nvidia Corporation/Personal AI Router` on macOS) plus bundled `manifests/*.json`. No database.
+- **Storage**: in-memory; manifests read from the per-user data dir's
+ `engines/*.json` (`%LocalAppData%\Nvidia Corporation\Personal AI Router` on
+ Windows, `~/.config/Nvidia Corporation/Personal AI Router` on Linux, and
+ `~/Library/Application Support/Nvidia Corporation/Personal AI Router` on
+ macOS) plus bundled `manifests/*.json`. llama.cpp uses a managed sibling cache
+ that survives uninstall and must currently be removed manually. No database.
## 10. Design Constraints
- **Performance**: control plane, not inference; sub-second RPCs except install (network-bound) and start (bounded by the readiness timeout).
@@ -194,7 +226,24 @@ The operator starts it: `engine:start {engine:"ollama"}` resolves the manifest r
The operator stops it: `engine:stop {engine:"ollama"}` signals a process the service owns. For an **adopted** engine (no owned process), it resolves the PID bound to the port and terminates it only when that process is running the binary we manage — reclaiming an orphan a prior run left on our own managed port; a genuinely foreign listener (a different image on a different port) is declined with an error naming its PID and image. A user-initiated `stop` records the OFF intent regardless (even when the RPC returns an error), so the health loop and restore-on-restart don't flip the engine back on — clients must not treat a stop error as proof the OFF choice was discarded. The cluster `ec` stop endpoint shares this semantics and may return HTTP 500 while OFF is persisted.
-The operator pulls a model: `engine:action {engine:"ollama", action:"pull_model", params:{name:"llama3.2"}}` issues the manifest-declared `POST 127.0.0.1:{port}/api/pull`. Because the action is `pull_model`, the request is routed through the streaming pull path (not the buffered `engine:action` reader): each `/api/pull` status line is emitted as an `engine:pull-progress` notification — so a local pull shows live download progress just like a remote pull's `engine:remote-progress` — and the request settles with the pull's terminal result line. Frames are coalesced (only a change in `stage` or `percent` is emitted) so a chatty engine that streams many byte-progress lines per layer doesn't flood subscribers. The engine's terminal `{"status":"success"}` surfaces as a `stage:"success"` frame; a **failed** pull emits a terminal `stage:"error", percent:-1, message:` frame in addition to the JSON-RPC error, so a UI whose synchronous call already timed out on a long download still converges off "pulling". A CLI-driven pull (LM Studio's `lms get`) has no line-level progress, so it emits one `stage:"pulling"` marker and returns the command's result. On `shutdown` (or stdin EOF) the service stops every running engine first, so none are orphaned.
+The operator pulls a model: `engine:action {engine:"ollama", action:"pull_model", params:{name:"llama3.2"}}` issues the manifest-declared `POST 127.0.0.1:{port}/api/pull`. Because the action is `pull_model`, the request is routed through the streaming pull path (not the buffered `engine:action` reader): each `/api/pull` status line is emitted as an `engine:pull-progress` notification — so a local pull shows live download progress just like a remote pull's `engine:remote-progress` — and the request settles with the pull's terminal result line. Frames are coalesced (only a change in `stage` or `percent` is emitted) so a chatty engine that streams many byte-progress lines per layer doesn't flood subscribers. Streaming Ollama and llama.cpp pulls use a 30-minute inactivity watchdog that is refreshed only when a layer/file completed-byte count advances, allowing active downloads to exceed 30 minutes without letting duplicate progress or heartbeat frames keep a stalled pull alive. The engine's terminal `{"status":"success"}` surfaces as a `stage:"success"` frame; a **failed** pull emits a terminal `stage:"error", percent:-1, message:` frame in addition to the JSON-RPC error, so a UI whose synchronous call already timed out on a long download still converges off "pulling". A CLI-driven pull (LM Studio's `lms get`) has no line-level progress, so it emits one `stage:"pulling"` marker and retains the fixed 30-minute action timeout before returning the command's result. On `shutdown` (or stdin EOF) the service stops every running engine first, so none are orphaned.
+
+For the `llamacpp-models-sse` adapter, the monitored model must match
+`params.model`. Subscribe before starting, and bound the `POST /models`
+acknowledgement by a separate 30-second total timeout that survives caller
+cancellation. Once accepted, any exit before a matching `download_finished` or
+`download_failed` event performs cleanup before returning: caller cancellation,
+remote disconnect, inactivity timeout, SSE read failure, and premature SSE EOF
+all take this path. Using the same captured router URL and a fresh five-second
+context, query `GET /models` and send `POST /models/unload` with the exact model
+only if its status is still `downloading`. Missing or completed models need no
+stop request, and persistent cache files are retained. Validate the unload
+acknowledgement (`success:true`); join cleanup failures to the original error
+without losing cancellation or inactivity causes. A lost or malformed start
+acknowledgement means acceptance and cancellation cannot be confirmed: do not
+unload a download without confirmed ownership. Neither terminal SSE event
+triggers cleanup. These operations retain the existing JSON-RPC and progress
+payloads and do not terminate the entire engine when cleanup fails.
## 15. Current integration / wiring
diff --git a/services/nvpair-job-scheduler/README.md b/services/nvpair-job-scheduler/README.md
index 63ffc434..e347a453 100644
--- a/services/nvpair-job-scheduler/README.md
+++ b/services/nvpair-job-scheduler/README.md
@@ -57,8 +57,8 @@ rank thrash. Invalid, missing, or older-than-10-second telemetry contributes a
neutral pressure of 1.
Nodes are sorted by `pending + gpuPressure`, then lower GPU pressure, then
-stable node ID. Pending counts include **both** engines together, so Ollama load
-affects the LM Studio ordering and vice versa.
+stable node ID. Pending counts include **all** engines together, so work through
+one facade affects every other facade's ordering.
Rankings are recomputed when the node set, catalog, or effective pressure
changes, and reconciled on the interval timer. A ranking is only emitted when
@@ -66,7 +66,8 @@ the order, pending counts, or pressure actually changed.
## Output
-One `schedule:priority` notification per engine (`ollama`, `lmstudio`):
+One `schedule:priority` notification per engine (`ollama`, `lmstudio`,
+`llamacpp`):
```json
{
@@ -84,7 +85,7 @@ One `schedule:priority` notification per engine (`ollama`, `lmstudio`):
```
The broker relays each snapshot to the matching proxy as `node/set-priority`.
-Both engines currently receive the same node-wide ordering; the per-engine
+All engines currently receive the same node-wide ordering; the per-engine
envelope exists so the routing contract can diverge later without a wire change.
Each proxy then adds its own reservations for in-flight requests whose workload
diff --git a/services/nvpair-job-scheduler/spec.md b/services/nvpair-job-scheduler/spec.md
index 9cab523a..ac87c007 100644
--- a/services/nvpair-job-scheduler/spec.md
+++ b/services/nvpair-job-scheduler/spec.md
@@ -289,9 +289,10 @@ Let `D` = current discovered-node `hostUuid`s:
`scheduledOn` (not `originatedFrom`) is the key: we care where work runs, not where
it came from. Engine remains part of workload identity and selects the downstream
-proxy output, but it does not partition the load metric: Ollama and LM Studio
-normally contend for the same node-level GPU, VRAM, CPU, and memory. Until resource
-affinity is observable, total node queue depth is the conservative signal.
+proxy output, but it does not partition the load metric: Ollama, LM Studio, and
+llama.cpp normally contend for the same node-level GPU, VRAM, CPU, and memory.
+Until resource affinity is observable, total node queue depth is the conservative
+signal.
### 7.3 Selection-state (proxy side, for reference)
@@ -417,7 +418,7 @@ The Broker spawns `nvpair-job-scheduler`, replays active workloads and telemetry
then sends discovery and resumes all three live streams. Discovery seeds
`GPU-RIG`, `MY-PC`, `LAB-DESK-B`. Pending counts are `3`, `0`, `1` and GPU
pressures are `3`, `0`, `1`, so combined loads are `6`, `0`, `2`. The scheduler
-emits `["MY-PC","LAB-DESK-B","GPU-RIG"]` for both engines. The ranking is
+emits `["MY-PC","LAB-DESK-B","GPU-RIG"]` for every engine. The ranking is
node-global, so the Broker coalesces the duplicate and delivers it once per
distinct proxy process via `node/set-priority`; each facade applies the subset it
can route to.
diff --git a/services/nvpair-proxy/README.md b/services/nvpair-proxy/README.md
index 20c753e1..87346832 100644
--- a/services/nvpair-proxy/README.md
+++ b/services/nvpair-proxy/README.md
@@ -22,7 +22,7 @@ because the broker plans a different port for each engine.
They share a process on purpose. Between scheduler snapshots a facade takes
short-lived reservations for work it has dispatched, and those live in the
process — a process per engine split that picture, so simultaneous bursts on
-both engines could pick the same node believing it idle.
+different engines could pick the same node believing it idle.
The cost is shared fate for the **process**: a crash takes every facade down and
the supervisor restarts them together. Smaller failures are contained to one
@@ -60,7 +60,7 @@ success, so a redelivered enable never tears down a working listener.
### Flags
Only process-scoped settings are flags. Anything per-engine is a `facade/enable`
-parameter, because one flag cannot carry two engines' plans.
+parameter, because one flag cannot carry multiple engines' plans.
| Flag | Default | Description |
|------|---------|-------------|
@@ -73,7 +73,7 @@ parameter, because one flag cannot carry two engines' plans.
| Field | Default | Description |
|-------|---------|-------------|
-| `engine` | *(required)* | Which engine to front: `ollama` or `lmstudio`. An unknown name is rejected with the accepted values. |
+| `engine` | *(required)* | Which engine to front: `ollama`, `lmstudio`, or `llamacpp`. An unknown name is rejected with the accepted values. |
| `port` | per engine, see below | HTTP listen port for request forwarding. Must be 1–65535, or omitted for the engine's standalone default. `0` means "the default" rather than "pick an ephemeral port", and any other out-of-range value is rejected, because the facade announces the requested port in its `ready` notification and the broker would be told `0`. |
| `aliasAddresses` | *(empty)* | Optional secondary `host:port` values for the same routing handler, one per loopback family so `localhost` resolves either way. Only literal loopback addresses are accepted; the broker uses this for a safe inherited local `OLLAMA_HOST`, and the aliases are not advertised to peers. Accepted only for an engine with an inherited host variable — today Ollama alone — and rejected for any other. |
| `ignorePersistedPort` | `false` | Use `port` even when a saved port exists (used by broker-managed startup) |
@@ -83,10 +83,10 @@ parameter, because one flag cannot carry two engines' plans.
Everything engine-specific is one entry in `engines.go`, plus the shared
identity in `nvpair-shared/engines`.
-| | `"engine":"ollama"` | `"engine":"lmstudio"` |
-|---|---|---|
-| Facade id — error-ID prefix, broker relay namespace, TUI proxies-view tab | `ollama-proxy` | `lmstudio-proxy` |
-| Discovery service key | `ol` | `lm` |
+| | `"engine":"ollama"` | `"engine":"lmstudio"` | `"engine":"llamacpp"` |
+|---|---|---|---|
+| Facade id — error-ID prefix and broker relay namespace | `ollama-proxy` | `lmstudio-proxy` | `llamacpp-proxy` |
+| Discovery service key | `ol` | `lm` | `lc` |
The **log component, supervisor label, and TUI health crash key are not in that
table**: they name the process (`nvpair-proxy`), not a facade, because one
@@ -95,25 +95,31 @@ Facade-scoped log records carry an `engine` field instead. The supervisor label
and the health crash key are matched against each other, so they move together
— see `nvpair-shared/engines` and `spec.md` §9.
-| | `"engine":"ollama"` | `"engine":"lmstudio"` |
-|---|---|---|
-| Engine's own client-facing port | 11434 | 1234 |
-| Where PAIR relocates the engine | 11435 | 1235 |
-| Standalone port, used when `port` is omitted | 11435 | 1234 |
-| Persisted-port file (declared, not derived) | `proxy-port.json` | `lmstudio-proxy-port.json` |
-| Model-list routes | `GET /api/tags` (native), `GET /v1/models` (OpenAI) | `GET /v1/models` (OpenAI) |
-| Inference routes | `/api/generate`, `/api/chat`, `/api/embeddings`, `/api/embed`, plus the OpenAI and Anthropic Messages sets | `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings`, `/v1/messages` |
-| Model naming | untagged means `:latest`, so `llama3` and `llama3:latest` are one model | identifiers compared byte for byte |
+| | `"engine":"ollama"` | `"engine":"lmstudio"` | `"engine":"llamacpp"` |
+|---|---|---|---|
+| Engine's own client-facing port | 11434 | 1234 | 8080 |
+| Where PAIR relocates the engine | 11435 | 1235 | 8081 |
+| Standalone port, used when `port` is omitted | 11435 | 1234 | 8080 |
+| Persisted-port file (declared, not derived) | `proxy-port.json` | `lmstudio-proxy-port.json` | `llamacpp-proxy-port.json` |
+| Model-list routes | `GET /api/tags` (native), `GET /v1/models` (OpenAI) | `GET /v1/models` (OpenAI) | `GET /models`, `GET /v1/models` (OpenAI; both query upstream `/models`) |
+| Inference routes | `/api/generate`, `/api/chat`, `/api/embeddings`, `/api/embed`, plus the OpenAI and Anthropic Messages sets | `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings`, `/v1/messages` | `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings` |
+| Model naming | untagged means `:latest`, so `llama3` and `llama3:latest` are one model | identifiers compared byte for byte | identifiers compared byte for byte |
The route table is a **classifier, not an allowlist**. An unlisted path is
forwarded verbatim, which is how `/api/show`, `/api/pull`, `/api/ps`,
`/api/version` and `OPTIONS` preflights keep working.
+The broker enables all three facades by default. Local llama.cpp-compatible
+clients use `8080`, while the managed `llama-server` stays on `8081`.
+`--proxy-engines` can restrict a standalone broker or TUI to a subset.
+
### HTTP Reverse Proxy
A facade listens on its enabled port and forwards incoming requests to the
currently active node — except the model-list routes, which are queried across
every candidate node concurrently and merged into one de-duplicated inventory.
+On the llama.cpp facade, `GET /models` and `GET /v1/models` return the same
+fleet inventory across llama.cpp candidates, even when a node is selected.
Point your client at the proxy and it handles routing.
When the broker supplies `aliasAddresses`, the facade reserves that
diff --git a/services/nvpair-proxy/engines.go b/services/nvpair-proxy/engines.go
index 71201a30..9ad19bab 100644
--- a/services/nvpair-proxy/engines.go
+++ b/services/nvpair-proxy/engines.go
@@ -78,8 +78,12 @@ const (
// route is one classified request path.
type route struct {
+ // Path is the client-facing path matched on the facade.
Path string
- Role routeRole
+
+ // UpstreamPath optionally rewrites Path for the engine; empty preserves it.
+ UpstreamPath string
+ Role routeRole
}
// engineProfile is everything the proxy needs to front one engine.
@@ -134,6 +138,13 @@ var lmStudioBaseRoutes = []route{
{Path: "/v1/models", Role: roleModelListOpenAIGET},
}
+// llamaCPPBaseRoutes serves both model-list paths through the fleet inventory,
+// querying llama.cpp's router endpoint. Its response uses the OpenAI envelope.
+var llamaCPPBaseRoutes = []route{
+ {Path: "/models", Role: roleModelListOpenAIGET},
+ {Path: "/v1/models", UpstreamPath: "/models", Role: roleModelListOpenAIGET},
+}
+
// openAIInferenceRoutes is the OpenAI-compatible inference surface.
var openAIInferenceRoutes = []route{
{Path: "/v1/chat/completions", Role: roleInferencePOST},
@@ -151,8 +162,10 @@ var profiles = buildProfiles()
func buildProfiles() []engineProfile {
ollama, _ := engines.ByName("ollama")
lmstudio, _ := engines.ByName("lmstudio")
+ llamacpp, _ := engines.ByName("llamacpp")
ollamaRoutes := slices.Concat(ollamaBaseRoutes, openAIInferenceRoutes, anthropicInferenceRoutes)
lmStudioRoutes := slices.Concat(lmStudioBaseRoutes, openAIInferenceRoutes, anthropicInferenceRoutes)
+ llamaCPPRoutes := slices.Concat(llamaCPPBaseRoutes, openAIInferenceRoutes)
return []engineProfile{
{
@@ -173,6 +186,13 @@ func buildProfiles() []engineProfile {
// stored value predates the current default of 1234.
ReservedPersistedPort: 1235,
},
+ {
+ Engine: llamacpp,
+ StandalonePort: 8080,
+ Routes: llamaCPPRoutes,
+ ModelNaming: exactID,
+ ReservedPersistedPort: 8081,
+ },
}
}
@@ -195,9 +215,9 @@ func engineNames() string {
return strings.Join(names, ", ")
}
-// roleFor classifies a request. The bool reports whether the path is one this
+// routeFor classifies a request. The bool reports whether the path is one this
// engine handles specially; false means forward it verbatim.
-func (p engineProfile) roleFor(method, path string) (routeRole, bool) {
+func (p engineProfile) routeFor(method, path string) (route, bool) {
for _, r := range p.Routes {
// Keep scanning on a method mismatch rather than bailing: a path may
// legitimately appear twice under different methods, and returning
@@ -206,9 +226,21 @@ func (p engineProfile) roleFor(method, path string) (routeRole, bool) {
if r.Path != path || r.Role.method() != method {
continue
}
- return r.Role, true
+ return r, true
+ }
+ return route{}, false
+}
+
+func (p engineProfile) roleFor(method, path string) (routeRole, bool) {
+ r, ok := p.routeFor(method, path)
+ return r.Role, ok
+}
+
+func (r route) upstreamPath() string {
+ if r.UpstreamPath != "" {
+ return r.UpstreamPath
}
- return 0, false
+ return r.Path
}
// method is the HTTP method a role applies to.
diff --git a/services/nvpair-proxy/engines_test.go b/services/nvpair-proxy/engines_test.go
index 422159c7..201a3046 100644
--- a/services/nvpair-proxy/engines_test.go
+++ b/services/nvpair-proxy/engines_test.go
@@ -3,7 +3,57 @@
package main
-import "testing"
+import (
+ "net/http"
+ "slices"
+ "testing"
+
+ "nvpair-shared/engines"
+)
+
+func TestProfilesMatchSharedEngines(t *testing.T) {
+ names := make([]string, len(profiles))
+ for i, profile := range profiles {
+ names[i] = profile.Name
+ }
+ if !slices.Equal(names, engines.Names()) {
+ t.Fatalf("proxy profiles = %v, want canonical engines %v", names, engines.Names())
+ }
+}
+
+func TestLlamaCPPProfile(t *testing.T) {
+ profile, ok := profileFor("llamacpp")
+ if !ok {
+ t.Fatal("llamacpp profile missing")
+ }
+ if role, ok := profile.roleFor("POST", "/v1/chat/completions"); !ok || role != roleInferencePOST {
+ t.Fatalf("chat route = %v, %v", role, ok)
+ }
+ if got := profile.normalizeModel("org/model:Q4_K_M"); got != "org/model:Q4_K_M" {
+ t.Fatalf("exact model id normalized to %q", got)
+ }
+ if profile.StandalonePort != 8080 || profile.ReservedPersistedPort != 8081 {
+ t.Fatalf("ports = facade %d, reserved %d", profile.StandalonePort, profile.ReservedPersistedPort)
+ }
+}
+
+func TestLlamaCPPModelListRoutes(t *testing.T) {
+ profile, ok := profileFor("llamacpp")
+ if !ok {
+ t.Fatal("llamacpp profile missing")
+ }
+ for _, path := range []string{"/models", "/v1/models"} {
+ t.Run(path, func(t *testing.T) {
+ route, ok := profile.routeFor(http.MethodGet, path)
+ if !ok {
+ t.Fatalf("GET %s is not classified", path)
+ }
+ if route.Role != roleModelListOpenAIGET || route.upstreamPath() != "/models" {
+ t.Fatalf("GET %s route = %+v, want OpenAI model list at upstream /models", path, route)
+ }
+ })
+ }
+}
// Routes is a classifier, not an allowlist. handlePlain forwards every
// loopback path into handleHTTP with no filtering, so a path the table does
@@ -18,6 +68,10 @@ func TestRoleForClassifiesOnlyDeclaredRoutes(t *testing.T) {
if !ok {
t.Fatal("lmstudio profile missing")
}
+ llamacpp, ok := profileFor("llamacpp")
+ if !ok {
+ t.Fatal("llamacpp profile missing")
+ }
for _, tc := range []struct {
name string
@@ -34,10 +88,12 @@ func TestRoleForClassifiesOnlyDeclaredRoutes(t *testing.T) {
{"ollama openai list", ollama, "GET", "/v1/models", roleModelListOpenAIGET, true},
{"ollama passthrough", ollama, "POST", "/api/pull", 0, false},
{"ollama version passthrough", ollama, "GET", "/api/version", 0, false},
+ {"ollama models passthrough", ollama, http.MethodGet, "/models", 0, false},
{"lmstudio chat", lmstudio, "POST", "/v1/chat/completions", roleInferencePOST, true},
{"lmstudio anthropic messages", lmstudio, "POST", "/v1/messages", roleInferencePOST, true},
{"lmstudio list", lmstudio, "GET", "/v1/models", roleModelListOpenAIGET, true},
+ {"lmstudio models passthrough", lmstudio, http.MethodGet, "/models", 0, false},
// LM Studio serves no native Ollama routes, so /api/chat is not
// inference for it — it is forwarded verbatim like any other path.
{"lmstudio has no native routes", lmstudio, "POST", "/api/chat", 0, false},
@@ -47,6 +103,7 @@ func TestRoleForClassifiesOnlyDeclaredRoutes(t *testing.T) {
// inference path would emit a workload.
{"wrong method on list", ollama, "POST", "/v1/models", 0, false},
{"wrong method on inference", ollama, "GET", "/api/chat", 0, false},
+ {"llamacpp models post passthrough", llamacpp, http.MethodPost, "/models", 0, false},
} {
t.Run(tc.name, func(t *testing.T) {
role, ok := tc.profile.roleFor(tc.method, tc.path)
diff --git a/services/nvpair-proxy/failover_test.go b/services/nvpair-proxy/failover_test.go
index 0d54b6fe..049e1d13 100644
--- a/services/nvpair-proxy/failover_test.go
+++ b/services/nvpair-proxy/failover_test.go
@@ -13,6 +13,7 @@ import (
"net/url"
"strconv"
"strings"
+ "sync/atomic"
"testing"
"time"
)
@@ -536,6 +537,91 @@ func TestHandleHTTP_AggregatesOpenAIModelList(t *testing.T) {
})
}
+func TestHandleHTTP_LlamaCPPModelListsAggregateFleet(t *testing.T) {
+ for _, path := range []string{"/models", "/v1/models"} {
+ t.Run(path, func(t *testing.T) {
+ profile, ok := profileFor("llamacpp")
+ if !ok {
+ t.Fatal("llamacpp profile missing")
+ }
+ serve := func(body string, hits *atomic.Int32) *httptest.Server {
+ t.Helper()
+ server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ hits.Add(1)
+ if r.Method != http.MethodGet || r.URL.Path != "/models" || r.URL.RawQuery != "scope=all" {
+ t.Errorf("upstream request = %s %s?%s, want GET /models?scope=all", r.Method, r.URL.Path, r.URL.RawQuery)
+ }
+ w.Header().Set("Content-Type", "application/json")
+ if _, err := io.WriteString(w, body); err != nil {
+ t.Errorf("write model list: %v", err)
+ }
+ }))
+ t.Cleanup(server.Close)
+ return server
+ }
+ var aHits, bHits atomic.Int32
+ a := serve(`{"object":"list","data":[{"id":"a","owned_by":"a"},{"id":"shared","owned_by":"first"}]}`, &aHits)
+ b := serve(`{"object":"list","data":[{"id":"shared","owned_by":"second"},{"id":"c","owned_by":"b"}]}`, &bHits)
+ disc := NewDiscovery()
+ disc.AddManual(nodeFor(t, "a", a.URL))
+ disc.AddManual(nodeFor(t, "b", b.URL))
+ f := testProxy(profile, disc, profile.FacadePort).soleFacade()
+ f.SetSelected("a")
+ rec := httptest.NewRecorder()
+ f.handleHTTP(rec, httptest.NewRequest(http.MethodGet, path+"?scope=all", nil))
+
+ if rec.Code != http.StatusOK {
+ t.Fatalf("model list status = %d, want %d", rec.Code, http.StatusOK)
+ }
+ var got struct {
+ Object string `json:"object"`
+ Data []struct {
+ ID string `json:"id"`
+ OwnedBy string `json:"owned_by"`
+ } `json:"data"`
+ }
+ if err := json.Unmarshal(rec.Body.Bytes(), &got); err != nil {
+ t.Fatalf("decode model list: %v", err)
+ }
+ if got.Object != "list" || len(got.Data) != 3 {
+ t.Fatalf("model list = %+v, want list envelope with three deduplicated models", got)
+ }
+ if got.Data[0].ID != "a" || got.Data[1].ID != "shared" || got.Data[1].OwnedBy != "first" || got.Data[2].ID != "c" {
+ t.Fatalf("models = %+v, want a, shared(first), c", got.Data)
+ }
+ if aHits.Load() != 1 || bHits.Load() != 1 {
+ t.Fatalf("upstream requests: a=%d, b=%d, want one per node despite selecting a", aHits.Load(), bHits.Load())
+ }
+ })
+ }
+}
+
+func TestHandleHTTP_ModelListRemapsUpstreamPath(t *testing.T) {
+ upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ if r.Method != http.MethodGet || r.URL.Path != "/models" || r.URL.RawQuery != "scope=all" {
+ t.Errorf("upstream request = %s %s?%s, want GET /models?scope=all", r.Method, r.URL.Path, r.URL.RawQuery)
+ }
+ _, _ = io.WriteString(w, `{"data":[{"id":"remapped"}]}`)
+ }))
+ defer upstream.Close()
+
+ profile := lmstudioCase(t).profile
+ profile.Routes = []route{{
+ Path: "/v1/models",
+ UpstreamPath: "/models",
+ Role: roleModelListOpenAIGET,
+ }}
+ disc := NewDiscovery()
+ disc.AddManual(nodeFor(t, "remapped", upstream.URL))
+ rec := httptest.NewRecorder()
+ testProxy(profile, disc, profile.FacadePort).soleFacade().
+ handleHTTP(rec, httptest.NewRequest(http.MethodGet, "/v1/models?scope=all", nil))
+
+ if rec.Code != http.StatusOK || !strings.Contains(rec.Body.String(), `"id":"remapped"`) {
+ t.Fatalf("response = %d %s, want remapped model list", rec.Code, rec.Body.String())
+ }
+}
+
func TestHandleHTTP_ModelListEmptyAndUnavailable(t *testing.T) {
forEachEngine(t, func(t *testing.T, tc engineCase) {
serveEmpty := func() *httptest.Server {
diff --git a/services/nvpair-proxy/proxy.go b/services/nvpair-proxy/proxy.go
index 929c3d67..52fb0f25 100644
--- a/services/nvpair-proxy/proxy.go
+++ b/services/nvpair-proxy/proxy.go
@@ -931,11 +931,11 @@ type modelListResult struct {
// in candidate order, not completion order, so duplicate metadata is
// deterministic while an unavailable peer cannot hide healthy inventories.
//
-// role selects the wire dialect: which array the upstream envelope carries,
-// which field identifies a record, and how the federated response is shaped.
-func (f *facade) serveModelList(w http.ResponseWriter, r *http.Request, role routeRole, candidates []candidate) (int, error) {
+// matchedRoute selects the wire dialect and the optional upstream path. The
+// client-facing path remains unchanged in telemetry and in the merged response.
+func (f *facade) serveModelList(w http.ResponseWriter, r *http.Request, matchedRoute route, candidates []candidate) (int, error) {
p := f.host
- openAI := role == roleModelListOpenAIGET
+ openAI := matchedRoute.Role == roleModelListOpenAIGET
writeJSON := func(status int, body []byte) {
w.Header().Set("Content-Type", "application/json")
w.Header().Set("X-Content-Type-Options", "nosniff")
@@ -946,8 +946,12 @@ func (f *facade) serveModelList(w http.ResponseWriter, r *http.Request, role rou
var wg sync.WaitGroup
for i, cand := range candidates {
target := *cand.url
- target.Path = r.URL.Path
- target.RawPath = r.URL.RawPath
+ target.Path = matchedRoute.upstreamPath()
+ if matchedRoute.UpstreamPath == "" {
+ target.RawPath = r.URL.RawPath
+ } else {
+ target.RawPath = ""
+ }
target.RawQuery = r.URL.RawQuery
upstream, err := http.NewRequestWithContext(r.Context(), http.MethodGet, target.String(), nil)
if err != nil {
@@ -1208,13 +1212,13 @@ func (f *facade) handleHTTP(w http.ResponseWriter, r *http.Request) {
p.releaseReservation(held)
held = reservation{}
}()
- if role, ok := f.profile.roleFor(r.Method, r.URL.Path); ok && role.isModelList() {
+ if matchedRoute, ok := f.profile.routeFor(r.Method, r.URL.Path); ok && matchedRoute.Role.isModelList() {
if len(candidates) > 0 {
_ = f.notify("proxy/request-started", RequestStartedEvent{
ID: reqID, Method: r.Method, Path: r.URL.Path, Target: "cluster",
})
}
- status, err := f.serveModelList(w, r, role, candidates)
+ status, err := f.serveModelList(w, r, matchedRoute, candidates)
errText := ""
if err != nil {
errText = err.Error()
diff --git a/services/nvpair-proxy/spec.md b/services/nvpair-proxy/spec.md
index 18c59999..0ab6722f 100644
--- a/services/nvpair-proxy/spec.md
+++ b/services/nvpair-proxy/spec.md
@@ -98,17 +98,17 @@ The process starts with **no engine and no listener**. The broker then sends one
`facade/enable` per engine, carrying that engine's port and any alias addresses.
A flag cannot express this. The broker plans a different port for each engine —
-Ollama's managed facade wants `:11434` while LM Studio's wants `:1234`, and
-either may be absent so the child keeps its own persisted port — and a
-single-valued flag carries only one plan.
+Ollama's managed facade wants `:11434`, LM Studio's wants `:1234`, and
+llama.cpp's managed facade wants `:8080`; any may be absent so the child keeps
+its own persisted port — and a single-valued flag carries only one plan.
### 3.1 Why one process
Between scheduler snapshots a facade takes short-lived **reservations** for work
it has dispatched but that the scheduler has not yet observed. Those live in the
process, and every facade shares them. If each engine had its own proxy
-process, neither could see the other's reservations, so simultaneous bursts of
-Ollama and LM Studio requests could both select the same node, each incorrectly
+process, none could see the others' reservations, so simultaneous bursts across
+Ollama, LM Studio, and llama.cpp could select the same node, each incorrectly
believing it was idle.
Sharing the map is the reason the engines share a process.
@@ -163,8 +163,8 @@ For a model-bearing inference request:
1. Filter a request-local discovery snapshot to nodes whose per-engine inventory
advertises the requested model. Ollama normalizes the implicit `:latest` tag;
- LM Studio ids match exactly. An empty owner set returns a local `502` without
- contacting an engine.
+ LM Studio and llama.cpp ids match exactly. An empty owner set returns a local
+ `502` without contacting an engine.
2. Order the eligible owners: explicit `node/select` pin, then the scheduler's
priority list, then deterministic default ordering.
3. Reserve the least estimated-loaded scheduler-listed candidate and move it to
@@ -178,6 +178,12 @@ For a model-bearing inference request:
An ineligible manual selection cannot override the capability gate, and failover
never broadens to an excluded node.
+The llama.cpp facade exposes `GET /models` and `GET /v1/models` as aliases for
+the fleet's llama.cpp inventory. Both query every candidate's `GET /models`
+concurrently and merge duplicate model ids into an OpenAI list envelope
+(`{"object":"list","data":[...]}`). A selected node affects candidate ordering,
+not inventory scope. Its inference routes remain `/v1/*`.
+
### 5.1 Retry bounds
| Bound | Value | Governs |
@@ -549,12 +555,12 @@ untouched, the primary listener stays up, and a warning is reported.
## 8. Ports
-| | Ollama | LM Studio |
-| --- | --- | --- |
-| Engine's own client-facing port | 11434 | 1234 |
-| Where PAIR relocates the engine | 11435 | 1235 |
-| Standalone port, when `port` is omitted | 11435 | 1234 |
-| Persisted-port file | `proxy-port.json` | `lmstudio-proxy-port.json` |
+| | Ollama | LM Studio | llama.cpp |
+| --- | --- | --- | --- |
+| Engine's own client-facing port | 11434 | 1234 | 8080 |
+| Where PAIR relocates the engine | 11435 | 1235 | 8081 |
+| Standalone port, when `port` is omitted | 11435 | 1234 | 8080 |
+| Persisted-port file | `proxy-port.json` | `lmstudio-proxy-port.json` | `llamacpp-proxy-port.json` |
A port chosen at runtime via `set-port` is persisted per engine and restored
when that facade is enabled, taking precedence over the requested port, so the
diff --git a/services/nvpair-tui/README.md b/services/nvpair-tui/README.md
index 3963d1a8..000f6d9c 100644
--- a/services/nvpair-tui/README.md
+++ b/services/nvpair-tui/README.md
@@ -30,9 +30,10 @@ Tabs:
| **Overview** | Broker liveness/version/uptime (`ping`) and a per-worker health table derived from the broker's `supervisor:subprocess-crashed:*` errors. |
| **Errors** | The service-error datastore (`errors:get-initial` + live `errors:update`); `c` clears the selected entry. |
| **Nodes** | mDNS-discovered Ollama nodes (`discovery:subscribe` / `discovery:nodes-changed`). |
-| **Proxies** | Ollama and LM Studio reverse proxies: status, discovered upstreams, select a node (`enter`/`a`), set the listen port (`p`). |
+| **Proxies** | Selected reverse-proxy facades: status, discovered upstreams, select a node (`enter`/`a`), set the listen port (`p`). Defaults to Ollama, LM Studio, and llama.cpp. |
| **Workloads** | Live cluster workloads (`workloads:subscribe` / `workloads:upsert` / `workloads:remove`). |
| **Engines** | Local inference engines: install (`i`), start (`s`), stop (`x`), restart (`r`), uninstall (`u`). |
+| **Models** | Local model inventory and loaded/idle/unknown state for every running engine; load (`enter`) or unload (`u`) the selected model. |
| **Cluster** | Pairing + membership: invite by address (`i`, shows the six-digit PIN — the first invite auto-founds a cluster of one), accept (`a`) / decline (`d`) an inbound invite, remove a member (`r`), leave (`L`). |
| **Manual** | User-added nodes: add by address (`a`), remove (`r`). |
| **Settings** | The node-settings store (force-ports, cluster auto-sync, cluster id/name). |
@@ -55,10 +56,15 @@ installed `bin/` layout). Override with `--broker-path`:
```sh
nvpair-tui # broker is a sibling binary
nvpair-tui --broker-path /opt/nvpair/bin/nvpair-ui-broker
+nvpair-tui --proxy-engines llamacpp # restrict facades when needed
nvpair-tui --log-level debug # own logging (to stderr)
nvpair-tui --version
```
+`--proxy-engines` accepts canonical engine ids from the shared engine table,
+passes the same selection to the broker, and builds the Proxies view from it.
+The default is `ollama,lmstudio,llamacpp`.
+
Logging goes to stderr (the broker's logs are shown inside the **Logs**
tab, not on the terminal), so it never corrupts the full-screen UI.
diff --git a/services/nvpair-tui/main.go b/services/nvpair-tui/main.go
index 4ee976f9..c754f40c 100644
--- a/services/nvpair-tui/main.go
+++ b/services/nvpair-tui/main.go
@@ -31,6 +31,7 @@ var Version = "dev"
func main() {
brokerPath := flag.String("broker-path", "", "path to nvpair-ui-broker binary (default: ./nvpair-ui-broker alongside this executable)")
+ proxyEnginesCSV := flag.String("proxy-engines", defaultProxyEngineCSV(), "comma-separated engines to front with a proxy")
showVersion := flag.Bool("version", false, "print version and exit")
resolveLevel := applog.RegisterFlag(nil, slog.LevelInfo)
flag.Parse()
@@ -42,6 +43,12 @@ func main() {
applog.Init("nvpair-tui", resolveLevel())
+ proxyEngines, err := parseProxyEngines(*proxyEnginesCSV)
+ if err != nil {
+ slog.Error("invalid --proxy-engines", "err", err)
+ os.Exit(2)
+ }
+
resolvedBroker, err := resolveBrokerPath(*brokerPath)
if err != nil {
slog.Error("cannot locate broker", "err", err)
@@ -61,7 +68,7 @@ func main() {
}
}()
- sup, err := Spawn(ctx, resolvedBroker)
+ sup, err := Spawn(ctx, resolvedBroker, proxyEngines)
if err != nil {
slog.Error("failed to start broker", "err", err)
os.Exit(1)
@@ -70,7 +77,7 @@ func main() {
// The broker's stderr (its logs plus every worker's, prefixed) is fed
// into the UI's Logs view rather than the terminal, so it never
// collides with the full-screen TUI on stdout.
- if err := ui.Run(sup.Client, sup.Stderr); err != nil {
+ if err := ui.Run(sup.Client, sup.Stderr, proxyEngines); err != nil {
slog.Error("ui error", "err", err)
}
diff --git a/services/nvpair-tui/proxyengines.go b/services/nvpair-tui/proxyengines.go
new file mode 100644
index 00000000..30f62616
--- /dev/null
+++ b/services/nvpair-tui/proxyengines.go
@@ -0,0 +1,49 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package main
+
+import (
+ "fmt"
+ "strings"
+
+ "nvpair-shared/engines"
+)
+
+func defaultProxyEngineCSV() string {
+ return strings.Join(engines.Names(), ",")
+}
+
+func proxyEngineCSV(selected []engines.Engine) string {
+ names := make([]string, len(selected))
+ for i, engine := range selected {
+ names[i] = engine.Name
+ }
+ return strings.Join(names, ",")
+}
+
+// parseProxyEngines narrows user input at the TUI boundary before the selected
+// engines are passed to both the broker and the Proxies view.
+func parseProxyEngines(csv string) ([]engines.Engine, error) {
+ seen := map[string]bool{}
+ var selected []engines.Engine
+ for _, raw := range strings.Split(csv, ",") {
+ name := strings.TrimSpace(raw)
+ if name == "" {
+ continue
+ }
+ engine, ok := engines.ByName(name)
+ if !ok {
+ return nil, fmt.Errorf(
+ "unknown engine %q; known engines are %s",
+ name,
+ strings.Join(engines.Names(), ", "),
+ )
+ }
+ if !seen[name] {
+ seen[name] = true
+ selected = append(selected, engine)
+ }
+ }
+ return selected, nil
+}
diff --git a/services/nvpair-tui/proxyengines_test.go b/services/nvpair-tui/proxyengines_test.go
new file mode 100644
index 00000000..fcef2300
--- /dev/null
+++ b/services/nvpair-tui/proxyengines_test.go
@@ -0,0 +1,45 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package main
+
+import (
+ "strings"
+ "testing"
+)
+
+func TestDefaultProxyEngineCSVUsesSharedDefaults(t *testing.T) {
+ if got, want := defaultProxyEngineCSV(), "ollama,lmstudio,llamacpp"; got != want {
+ t.Fatalf("default proxy engines = %q, want %q", got, want)
+ }
+}
+
+func TestParseProxyEnginesPreservesExplicitSelection(t *testing.T) {
+ got, err := parseProxyEngines("llamacpp, ollama, llamacpp")
+ if err != nil {
+ t.Fatalf("parse explicit engines: %v", err)
+ }
+ if len(got) != 2 || got[0].Name != "llamacpp" || got[1].Name != "ollama" {
+ t.Fatalf("parsed engines = %v, want [llamacpp ollama]", got)
+ }
+}
+
+func TestParseProxyEnginesAllowsNoFacades(t *testing.T) {
+ got, err := parseProxyEngines(" , ")
+ if err != nil {
+ t.Fatalf("parse empty selection: %v", err)
+ }
+ if len(got) != 0 {
+ t.Fatalf("parsed engines = %v, want none", got)
+ }
+}
+
+func TestParseProxyEnginesRejectsUnknownEngine(t *testing.T) {
+ _, err := parseProxyEngines("llamacpp,vllm")
+ if err == nil {
+ t.Fatal("unknown engine was accepted")
+ }
+ if !strings.Contains(err.Error(), `unknown engine "vllm"`) {
+ t.Fatalf("error = %q, want unknown-engine detail", err)
+ }
+}
diff --git a/services/nvpair-tui/supervisor.go b/services/nvpair-tui/supervisor.go
index 546a118d..41532210 100644
--- a/services/nvpair-tui/supervisor.go
+++ b/services/nvpair-tui/supervisor.go
@@ -13,6 +13,7 @@ import (
"runtime"
"time"
+ "nvpair-shared/engines"
"nvpair-tui/rpc"
)
@@ -77,8 +78,8 @@ func resolveBrokerPath(override string) (string, error) {
// runs with its working directory set to the broker's own directory so
// the broker's sibling-binary worker resolution finds nvpair-node-scanner et
// al. ctx governs the client read loop; use Shutdown for an orderly stop.
-func Spawn(ctx context.Context, brokerPath string) (*Supervisor, error) {
- cmd := exec.Command(brokerPath)
+func Spawn(ctx context.Context, brokerPath string, proxyEngines []engines.Engine) (*Supervisor, error) {
+ cmd := exec.Command(brokerPath, brokerArgs(proxyEngines)...)
cmd.Dir = filepath.Dir(brokerPath)
configureSubprocess(cmd)
@@ -105,6 +106,10 @@ func Spawn(ctx context.Context, brokerPath string) (*Supervisor, error) {
return &Supervisor{cmd: cmd, stdin: stdin, Client: client, Stderr: stderr}, nil
}
+func brokerArgs(proxyEngines []engines.Engine) []string {
+ return []string{"--proxy-engines", proxyEngineCSV(proxyEngines)}
+}
+
// Shutdown asks the broker to stop cleanly: send the shutdown RPC, close
// its stdin (a second, EOF-based stop signal), then wait up to
// shutdownGrace before killing it. The broker tears its own workers down
diff --git a/services/nvpair-tui/supervisor_test.go b/services/nvpair-tui/supervisor_test.go
index ab039b31..5a131ad2 100644
--- a/services/nvpair-tui/supervisor_test.go
+++ b/services/nvpair-tui/supervisor_test.go
@@ -12,6 +12,8 @@ import (
"path/filepath"
"testing"
"time"
+
+ "nvpair-shared/engines"
)
// TestMain doubles as a fake nvpair-ui-broker when NVPAIR_TUI_FAKE_BROKER=1.
@@ -72,7 +74,7 @@ func TestSupervisorReadyAndShutdown(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
- sup, err := Spawn(ctx, os.Args[0])
+ sup, err := Spawn(ctx, os.Args[0], engines.All())
if err != nil {
t.Fatalf("spawn: %v", err)
}
@@ -105,3 +107,14 @@ func TestSupervisorReadyAndShutdown(t *testing.T) {
t.Fatal("shutdown did not complete")
}
}
+
+func TestBrokerArgsForwardProxySelection(t *testing.T) {
+ llamacpp, ok := engines.ByName("llamacpp")
+ if !ok {
+ t.Fatal("shared engine table has no llamacpp")
+ }
+ got := brokerArgs([]engines.Engine{llamacpp})
+ if len(got) != 2 || got[0] != "--proxy-engines" || got[1] != "llamacpp" {
+ t.Fatalf("broker args = %v, want [--proxy-engines llamacpp]", got)
+ }
+}
diff --git a/services/nvpair-tui/ui/engines.go b/services/nvpair-tui/ui/engines.go
index 809febfb..695e6a70 100644
--- a/services/nvpair-tui/ui/engines.go
+++ b/services/nvpair-tui/ui/engines.go
@@ -218,10 +218,9 @@ func (v *enginesView) handleKey(msg tea.KeyMsg) tea.Cmd {
// pullParams builds the engine:action{action:"pull_model"} params for a pull.
// The model name is sent under BOTH "name" and "model" — mirroring
-// PullModelStream's own empty-params default — because the two engines key it
-// differently: Ollama's pull_model is HTTP /api/pull (body key "name"), while
-// LM Studio's is a CLI action `lms get {model}` resolved from the "model" key.
-// Sending only one key silently no-ops the pull on the other engine.
+// PullModelStream's own empty-params default — because bundled engines key it
+// differently: Ollama's HTTP action uses "name", while LM Studio and llama.cpp
+// resolve "model". Sending only one key silently no-ops a supported engine.
func pullParams(engine, model string) map[string]any {
return map[string]any{"engine": engine, "action": "pull_model", "params": map[string]string{"name": model, "model": model}}
}
diff --git a/services/nvpair-tui/ui/models.go b/services/nvpair-tui/ui/models.go
new file mode 100644
index 00000000..0a75b6ff
--- /dev/null
+++ b/services/nvpair-tui/ui/models.go
@@ -0,0 +1,228 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package ui
+
+import (
+ "slices"
+
+ "nvpair-tui/rpc"
+
+ "github.com/charmbracelet/bubbles/key"
+ "github.com/charmbracelet/bubbles/table"
+ tea "github.com/charmbracelet/bubbletea"
+)
+
+type modelInventory struct {
+ ByEngine map[string][]string `json:"modelsByEngine"`
+ LoadedByEngine map[string][]string `json:"loadedByEngine"`
+}
+type modelRow struct {
+ engine, model, state string
+}
+type modelsView struct {
+ client *rpc.Client
+ table table.Model
+ rows []modelRow
+ status string
+}
+type modelsLoadedMsg struct {
+ inventory modelInventory
+ err error
+}
+type modelActionMsg struct {
+ what string
+ row modelRow
+ err error
+}
+type modelActionRequest struct {
+ Engine string `json:"engine"`
+ Action string `json:"action"`
+ Params modelActionParams `json:"params"`
+}
+type modelActionParams struct {
+ Model string `json:"model"`
+ Stream *bool `json:"stream,omitempty"`
+ KeepAlive *int `json:"keep_alive,omitempty"`
+}
+
+var (
+ modelLoadKey = key.NewBinding(key.WithKeys("enter"), key.WithHelp("enter", "load"))
+ modelUnloadKey = key.NewBinding(key.WithKeys("u"), key.WithHelp("u", "unload"))
+)
+
+func newModelsView(client *rpc.Client) *modelsView {
+ view := &modelsView{client: client, table: newTable(nil)}
+ view.SetSize(80, 20)
+ return view
+}
+func (v *modelsView) Title() string { return "Models" }
+
+func (v *modelsView) Init() tea.Cmd {
+ return tea.Batch(
+ call(v.client, "engine:subscribe", nil, func(_ *rpc.Message, _ error) tea.Msg { return nil }),
+ v.loadCmd(),
+ )
+}
+
+func (v *modelsView) loadCmd() tea.Cmd {
+ return call(v.client, "engine:models", nil, func(msg *rpc.Message, err error) tea.Msg {
+ if err != nil {
+ return modelsLoadedMsg{err: err}
+ }
+ var inventory modelInventory
+ err = decodeParams(msg.Result, &inventory)
+ return modelsLoadedMsg{inventory: inventory, err: err}
+ })
+}
+
+func (v *modelsView) SetSize(width, height int) {
+ const engineWidth, stateWidth = 14, 8
+ v.table.SetColumns([]table.Column{
+ {Title: "ENGINE", Width: engineWidth},
+ {Title: "MODEL", Width: clampWidth(width-engineWidth-stateWidth-2, 16)},
+ {Title: "STATE", Width: stateWidth},
+ })
+ v.table.SetWidth(width)
+ v.table.SetHeight(clampWidth(height-2, 1))
+}
+
+func (v *modelsView) Update(msg tea.Msg) tea.Cmd {
+ switch msg := msg.(type) {
+ case modelsLoadedMsg:
+ if msg.err != nil {
+ v.status = "load models failed: " + msg.err.Error()
+ } else {
+ v.status = ""
+ v.apply(msg.inventory)
+ }
+ return nil
+ case modelActionMsg:
+ if msg.err != nil {
+ v.status = msg.what + " " + msg.row.engine + "/" + msg.row.model + " failed: " + msg.err.Error()
+ } else {
+ v.status = msg.what + " " + msg.row.engine + "/" + msg.row.model + " ok"
+ }
+ return nil
+ case NotificationMsg:
+ if msg.Msg.Method == "engine:state-changed" {
+ return v.loadCmd()
+ }
+ if msg.Msg.Method != "engine:models-changed" {
+ return nil
+ }
+ var changed struct {
+ Models modelInventory `json:"models"`
+ }
+ if err := decodeParams(msg.Msg.Params, &changed); err != nil {
+ v.status = "update models failed: " + err.Error()
+ } else {
+ v.status = ""
+ v.apply(changed.Models)
+ }
+ return nil
+ case tea.KeyMsg:
+ return v.handleKey(msg)
+ }
+ return nil
+}
+
+func (v *modelsView) handleKey(msg tea.KeyMsg) tea.Cmd {
+ what := ""
+ switch {
+ case key.Matches(msg, modelLoadKey):
+ what = "load"
+ case key.Matches(msg, modelUnloadKey):
+ what = "unload"
+ default:
+ var cmd tea.Cmd
+ v.table, cmd = v.table.Update(msg)
+ return cmd
+ }
+ row, ok := v.selectedModel()
+ if !ok {
+ return nil
+ }
+ v.status = what + " " + row.engine + "/" + row.model + "..."
+ request := newModelActionRequest(row, what)
+ return call(v.client, "engine:action", request, func(_ *rpc.Message, err error) tea.Msg {
+ return modelActionMsg{what: what, row: row, err: err}
+ })
+}
+
+func newModelActionRequest(row modelRow, what string) modelActionRequest {
+ request := modelActionRequest{
+ Engine: row.engine,
+ Action: what + "_model",
+ Params: modelActionParams{Model: row.model},
+ }
+ if row.engine != "ollama" {
+ return request
+ }
+ switch what {
+ case "load":
+ stream := false
+ request.Action = "run_model"
+ request.Params.Stream = &stream
+ case "unload":
+ keepAlive := 0
+ request.Params.KeepAlive = &keepAlive
+ }
+ return request
+}
+
+func (v *modelsView) selectedModel() (modelRow, bool) {
+ index := v.table.Cursor()
+ if index < 0 || index >= len(v.rows) {
+ return modelRow{}, false
+ }
+ return v.rows[index], true
+}
+
+func (v *modelsView) apply(inventory modelInventory) {
+ engines := make([]string, 0, len(inventory.ByEngine))
+ for engine := range inventory.ByEngine {
+ engines = append(engines, engine)
+ }
+ slices.Sort(engines)
+
+ var rows []modelRow
+ var tableRows []table.Row
+ for _, engine := range engines {
+ models := append([]string(nil), inventory.ByEngine[engine]...)
+ slices.Sort(models)
+ for _, model := range models {
+ loadedModels, known := inventory.LoadedByEngine[engine]
+ state := "unknown"
+ if known {
+ state = "idle"
+ if slices.Contains(loadedModels, model) {
+ state = "loaded"
+ }
+ }
+ row := modelRow{engine: engine, model: model, state: state}
+ rows = append(rows, row)
+ tableRows = append(tableRows, table.Row{engine, model, state})
+ }
+ }
+ v.rows = rows
+ v.table.SetRows(tableRows)
+}
+
+func (v *modelsView) View() string {
+ if len(v.rows) == 0 {
+ if v.status != "" {
+ return statusErrStyle.Render(v.status)
+ }
+ return footerStyle.Render("No local models reported by running engines.")
+ }
+ out := v.table.View()
+ if v.status != "" {
+ out += "\n" + footerStyle.Render(v.status)
+ }
+ return out
+}
+
+func (v *modelsView) Help() []key.Binding {
+ return []key.Binding{modelLoadKey, modelUnloadKey}
+}
diff --git a/services/nvpair-tui/ui/models_test.go b/services/nvpair-tui/ui/models_test.go
new file mode 100644
index 00000000..a0956546
--- /dev/null
+++ b/services/nvpair-tui/ui/models_test.go
@@ -0,0 +1,93 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package ui
+
+import (
+ "errors"
+ "strings"
+ "testing"
+
+ "nvpair-tui/rpc"
+)
+
+func TestModelsViewLoadsEveryEngineAndResidency(t *testing.T) {
+ var inventory modelInventory
+ raw := []byte(`{"modelsByEngine":{"ollama":["b:latest","a:latest"],"llamacpp":["owner/model-GGUF:Q4_K_M"]},"loadedByEngine":{"ollama":["b:latest"],"llamacpp":["owner/model-GGUF:Q4_K_M"]}}`)
+ if err := decodeParams(raw, &inventory); err != nil {
+ t.Fatalf("decode baseline: %v", err)
+ }
+ view := newModelsView(nil)
+ view.Update(modelsLoadedMsg{inventory: inventory})
+ if len(view.rows) != 3 {
+ t.Fatalf("model rows = %v, want three rows", view.rows)
+ }
+ if row := view.rows[0]; row.engine != "llamacpp" || row.model != "owner/model-GGUF:Q4_K_M" || row.state != "loaded" {
+ t.Errorf("first row = %+v, want loaded llama.cpp model", row)
+ }
+ if row := view.rows[1]; row.engine != "ollama" || row.model != "a:latest" || row.state != "idle" {
+ t.Errorf("second row = %+v, want idle Ollama a:latest", row)
+ }
+ if row := view.rows[2]; row.model != "b:latest" || row.state != "loaded" {
+ t.Errorf("third row = %+v, want loaded Ollama b:latest", row)
+ }
+ view.table.SetCursor(1)
+ selected, ok := view.selectedModel()
+ if !ok || selected != view.rows[1] {
+ t.Fatalf("selected model = %+v, %v; want second row", selected, ok)
+ }
+}
+
+func TestModelsChangedReplacesResidencySnapshot(t *testing.T) {
+ view := newModelsView(nil)
+ view.apply(modelInventory{
+ ByEngine: map[string][]string{"llamacpp": {"owner/model:Q4_K_M"}},
+ LoadedByEngine: map[string][]string{"llamacpp": {}},
+ })
+ view.Update(NotificationMsg{Msg: &rpc.Message{
+ Method: "engine:models-changed",
+ Params: []byte(`{"engine":"llamacpp","models":{"modelsByEngine":{"llamacpp":["owner/model:Q4_K_M"]},"loadedByEngine":{"llamacpp":["owner/model:Q4_K_M"]}}}`),
+ }})
+ if len(view.rows) != 1 || view.rows[0].state != "loaded" {
+ t.Fatalf("rows after residency push = %+v, want one loaded model", view.rows)
+ }
+}
+
+func TestModelActionRequestsMatchEngineContracts(t *testing.T) {
+ test := func(name, engine, what, wantAction string, wantStream, wantKeepAlive bool) {
+ t.Run(name, func(t *testing.T) {
+ request := newModelActionRequest(modelRow{engine: engine, model: "owner/model"}, what)
+ if request.Engine != engine || request.Action != wantAction || request.Params.Model != "owner/model" {
+ t.Fatalf("request = %+v, want %s %s for owner/model", request, engine, wantAction)
+ }
+ if (request.Params.Stream != nil) != wantStream {
+ t.Errorf("stream presence = %v, want %v", request.Params.Stream != nil, wantStream)
+ }
+ if request.Params.Stream != nil && *request.Params.Stream {
+ t.Error("Ollama load must send stream=false")
+ }
+ if (request.Params.KeepAlive != nil) != wantKeepAlive {
+ t.Errorf("keep_alive presence = %v, want %v", request.Params.KeepAlive != nil, wantKeepAlive)
+ }
+ if request.Params.KeepAlive != nil && *request.Params.KeepAlive != 0 {
+ t.Errorf("keep_alive = %d, want 0", *request.Params.KeepAlive)
+ }
+ })
+ }
+ test("llama.cpp load", "llamacpp", "load", "load_model", false, false)
+ test("llama.cpp unload", "llamacpp", "unload", "unload_model", false, false)
+ test("Ollama load", "ollama", "load", "run_model", true, false)
+ test("Ollama unload", "ollama", "unload", "unload_model", false, true)
+}
+
+func TestModelsViewEmptyAndErrorStates(t *testing.T) {
+ view := newModelsView(nil)
+ if got := view.View(); !strings.Contains(got, "No local models") {
+ t.Fatalf("empty view = %q, want empty-state guidance", got)
+ }
+
+ view.Update(modelsLoadedMsg{err: errors.New("manager unavailable")})
+ if got := view.View(); !strings.Contains(got, "manager unavailable") {
+ t.Fatalf("error view = %q, want backend error", got)
+ }
+}
diff --git a/services/nvpair-tui/ui/proxies.go b/services/nvpair-tui/ui/proxies.go
index 42c36dca..c1986cb3 100644
--- a/services/nvpair-tui/ui/proxies.go
+++ b/services/nvpair-tui/ui/proxies.go
@@ -25,13 +25,11 @@ type proxyNode struct {
Port int `json:"port"`
}
-// buildProxyEngines makes one tab per engine, in the shared table's order, so
-// an engine added there appears here rather than being silently absent from
-// this view.
-func buildProxyEngines() []*proxyEngine {
- all := engines.All()
- out := make([]*proxyEngine, 0, len(all))
- for _, e := range all {
+// buildProxyEngines makes one tab per facade selected at the TUI boundary, in
+// the same order passed to the broker.
+func buildProxyEngines(selected []engines.Engine) []*proxyEngine {
+ out := make([]*proxyEngine, 0, len(selected))
+ for _, e := range selected {
out = append(out, &proxyEngine{label: e.DisplayName, prefix: e.ComponentName(), table: newTable(nil)})
}
return out
@@ -40,8 +38,8 @@ func buildProxyEngines() []*proxyEngine {
// proxyEngine is one reverse proxy the broker fronts. They all speak the same
// routing/failover contract; only the JSON-RPC prefix and label differ.
type proxyEngine struct {
- label string // "Ollama" / "LM Studio"
- prefix string // "ollama-proxy" / "lmstudio-proxy"
+ label string
+ prefix string
ready bool
port int
selected string
@@ -49,7 +47,7 @@ type proxyEngine struct {
table table.Model
}
-// proxiesView shows both reverse proxies: per-engine status (ready/port/
+// proxiesView shows the selected reverse proxies: per-engine status (ready/port/
// selected node) and the focused engine's discovered upstreams, with
// actions to select a node and set the listen port.
type proxiesView struct {
@@ -94,14 +92,14 @@ var (
proxyAutoKey = key.NewBinding(key.WithKeys("a"), key.WithHelp("a", "auto-select"))
)
-func newProxiesView(client *rpc.Client) *proxiesView {
+func newProxiesView(client *rpc.Client, selected []engines.Engine) *proxiesView {
ti := textinput.New()
ti.Placeholder = "port"
ti.CharLimit = 5
v := &proxiesView{
client: client,
portInput: ti,
- engines: buildProxyEngines(),
+ engines: buildProxyEngines(selected),
}
return v
}
diff --git a/services/nvpair-tui/ui/proxies_test.go b/services/nvpair-tui/ui/proxies_test.go
new file mode 100644
index 00000000..897d9597
--- /dev/null
+++ b/services/nvpair-tui/ui/proxies_test.go
@@ -0,0 +1,67 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package ui
+
+import (
+ "testing"
+
+ "nvpair-shared/engines"
+ "nvpair-tui/rpc"
+)
+
+func TestBuildProxyEnginesUsesSelectedEngines(t *testing.T) {
+ test := func(name string, selected []engines.Engine) {
+ t.Run(name, func(t *testing.T) {
+ got := buildProxyEngines(selected)
+ if len(got) != len(selected) {
+ t.Fatalf("proxy tabs = %d, want %d", len(got), len(selected))
+ }
+ for i, engine := range selected {
+ if got[i].label != engine.DisplayName || got[i].prefix != engine.ComponentName() {
+ t.Errorf("proxy tab %d = (%q, %q), want (%q, %q)",
+ i, got[i].label, got[i].prefix, engine.DisplayName, engine.ComponentName())
+ }
+ }
+ })
+ }
+
+ test("all shared engines", engines.All())
+ llamacpp, ok := engines.ByName("llamacpp")
+ if !ok {
+ t.Fatal("shared engine table has no llamacpp")
+ }
+ test("explicit llama.cpp", []engines.Engine{llamacpp})
+}
+
+func TestLlamaCPPNotificationsUseSelectedFacade(t *testing.T) {
+ llamacpp, ok := engines.ByName("llamacpp")
+ if !ok {
+ t.Fatal("shared engine table has no llamacpp")
+ }
+ view := newProxiesView(nil, []engines.Engine{llamacpp})
+
+ cmd := view.handleNotification(&rpc.Message{
+ Method: "llamacpp-proxy:ready",
+ Params: []byte(`{"port":8080}`),
+ })
+ if cmd != nil {
+ t.Fatal("ready notification unexpectedly returned a command")
+ }
+ if !view.engines[0].ready || view.engines[0].port != 8080 {
+ t.Fatalf("llama.cpp status = ready:%v port:%d, want ready on 8080",
+ view.engines[0].ready, view.engines[0].port)
+ }
+
+ if cmd := view.handleNotification(&rpc.Message{Method: "llamacpp-proxy:node/discovered"}); cmd == nil {
+ t.Fatal("llama.cpp node notification did not schedule a nodes refresh")
+ }
+}
+
+func TestDefaultViewsOmitProxiesWhenNoneSelected(t *testing.T) {
+ for _, view := range defaultViews(nil, nil) {
+ if view.Title() == "Proxies" {
+ t.Fatal("Proxies view present with no selected proxy engines")
+ }
+ }
+}
diff --git a/services/nvpair-tui/ui/ui.go b/services/nvpair-tui/ui/ui.go
index 65015a7b..02c2b9af 100644
--- a/services/nvpair-tui/ui/ui.go
+++ b/services/nvpair-tui/ui/ui.go
@@ -7,6 +7,7 @@ import (
"bufio"
"io"
+ "nvpair-shared/engines"
"nvpair-tui/rpc"
tea "github.com/charmbracelet/bubbletea"
@@ -15,12 +16,12 @@ import (
// Run builds the tabbed program over a connected broker client and the
// broker's stderr stream, and blocks until the user quits. The caller is
// responsible for shutting the broker down afterwards.
-func Run(client *rpc.Client, stderr io.Reader) error {
+func Run(client *rpc.Client, stderr io.Reader, proxyEngines []engines.Engine) error {
logCh := make(chan string, 2000)
go scanLines(stderr, logCh)
p := tea.NewProgram(
- New(client, logCh, defaultViews(client)),
+ New(client, logCh, defaultViews(client, proxyEngines)),
tea.WithAltScreen(),
)
_, err := p.Run()
@@ -40,17 +41,22 @@ func scanLines(r io.Reader, out chan<- string) {
}
// defaultViews lists the tabs in display order.
-func defaultViews(client *rpc.Client) []View {
- return []View{
+func defaultViews(client *rpc.Client, proxyEngines []engines.Engine) []View {
+ views := []View{
newHealthView(client),
newErrorsView(client),
newNodesView(client),
- newProxiesView(client),
+ }
+ if len(proxyEngines) > 0 {
+ views = append(views, newProxiesView(client, proxyEngines))
+ }
+ return append(views,
newWorkloadsView(client),
newEnginesView(client),
+ newModelsView(client),
newClusterView(client),
newManualView(client),
newSettingsView(client),
newLogsView(client),
- }
+ )
}
diff --git a/services/nvpair-ui-broker/ENGINE_SETTINGS.md b/services/nvpair-ui-broker/ENGINE_SETTINGS.md
index a6e2351f..5d5f5728 100644
--- a/services/nvpair-ui-broker/ENGINE_SETTINGS.md
+++ b/services/nvpair-ui-broker/ENGINE_SETTINGS.md
@@ -5,10 +5,10 @@ SPDX-License-Identifier: Apache-2.0
# Engine settings protocol
-The broker owns combined settings operations for Ollama and LM Studio. Each
-node owns its own configuration. `nodeId` selects a discovered, currently pinned
-peer; omission or the local host ID selects this node. Bulk propagation is not
-part of this API.
+The broker owns combined settings operations for Ollama, LM Studio, and
+llama.cpp. Each node owns its own configuration. `nodeId` selects a discovered,
+currently pinned peer; omission or the local host ID selects this node. Bulk
+propagation is not part of this API.
| Method | Request | Result |
| --- | --- | --- |
@@ -16,10 +16,10 @@ part of this API.
| `engine:preview-settings` | `{engine, nodeId?, expectedRevision, settings, resolution?}` | Normalized settings, errors, conflict, restart/rebind summary |
| `engine:apply-settings` | Preview request plus `requestId` | `{revision, phase}` acknowledgement |
-`engine` is `ollama` or `lmstudio`. `settings` contains all three fields:
-`serverPort`, `proxyPort`, `launchText`. The last field contains arguments and
-leading environment assignments, without the executable or startup subcommand.
-The argument grammar is
+`engine` is `ollama`, `lmstudio`, or `llamacpp`. `settings` contains all three
+fields: `serverPort`, `proxyPort`, `launchText`. The last field contains
+arguments and leading environment assignments, without the executable or
+startup subcommand. The argument grammar is
[`pair-arguments-v1`](../nvpair-engine-manager/LAUNCH_TEXT.md). Preview does not
change component configuration or runtime. A snapshot read may persist the
initial revision baseline or reconcile an external component change.
diff --git a/services/nvpair-ui-broker/README.md b/services/nvpair-ui-broker/README.md
index 91cd7f91..c875b6b6 100644
--- a/services/nvpair-ui-broker/README.md
+++ b/services/nvpair-ui-broker/README.md
@@ -23,7 +23,7 @@ namespace:
| --- | --- | --- |
| `nvpair-node-scanner` | Discovery daemon: advertises this host's one `_nvpair-node._tcp` record and browses the LAN | `discovery:*` |
| `nvpair-node-info` | Local GPU / CPU / memory inventory over HTTP at `/v1/node-info` | — (HTTP only) |
-| `nvpair-proxy` | One process hosting an inference proxy and router facade per enabled engine | `ollama-proxy:*`, `lmstudio-proxy:*` |
+| `nvpair-proxy` | One process hosting an inference proxy and router facade per enabled engine | `ollama-proxy:*`, `lmstudio-proxy:*`, `llamacpp-proxy:*` |
| `nvpair-engine-manager` | Local engine and model control plane; also serves `GET /v1/models` to peers | `engine:*` |
| `nvpair-cluster-manager` | Node identity, trusted-node store, PIN pairing | `cluster:*`, `nodes:*` |
| `nvpair-workload-manager` | Cluster workload relay between this node and peers | `workloads:*` |
@@ -39,9 +39,10 @@ lifecycle, and relay rules.
Two responsibilities live in the broker itself rather than in a worker:
-- **Engine advertising.** The broker polls local Ollama and LM Studio every 5 s
- and registers each running engine's port (`ol` / `lm`) with the discovery
- daemon, so both are carried in this host's single `_nvpair-node` record. The
+- **Engine advertising.** The broker polls local Ollama, LM Studio, and
+ llama.cpp every 5 s and registers each running engine's promoted facade port
+ (`ol` / `lm` / `lc`) with the discovery daemon, so they are carried in this
+ host's single `_nvpair-node` record. The
model list is not part of that record — it is served over HTTP by
`nvpair-engine-manager` on the `em` service and fetched by a peer's daemon
during discovery enrichment.
@@ -73,7 +74,7 @@ Bidirectional newline-delimited JSON-RPC 2.0 — same conventions as every other
| `--scanner-path ` | `./nvpair-node-scanner[.exe]` in the CWD | Explicit path to the `nvpair-node-scanner` binary the broker should spawn |
| `--node-info-path ` | `./nvpair-node-info[.exe]` in the CWD | Explicit path to the `nvpair-node-info` binary the broker should spawn. When omitted and no default sibling exists, the broker runs without the local inventory server (non-fatal); when set to an invalid path, the broker exits with an error |
| `--proxy-path ` | `./nvpair-proxy[.exe]` in the CWD | Explicit path to the `nvpair-proxy` binary. One process fronts every engine: the broker spawns it once and then sends a `facade/enable` per entry in `--proxy-engines`. Same optional semantics as `--node-info-path`: an absent default sibling means no local proxies (non-fatal); an invalid explicit path exits with an error |
-| `--proxy-engines ` | every engine in `nvpair-shared/engines` (currently `ollama,lmstudio`) | Which engines to front with a proxy. An unrecognized name exits with an error rather than being skipped, so a typo cannot look like it worked. An engine left out is not started **and not prepared** — the broker will not relocate an engine whose facade nothing is going to claim |
+| `--proxy-engines ` | `ollama,lmstudio,llamacpp` | Which engines to front with a proxy. An unrecognized name exits with an error rather than being skipped, so a typo cannot look like it worked. An engine left out is not started **and not prepared** — the broker will not relocate an engine whose facade nothing is going to claim |
| `--workload-manager-path ` | `./nvpair-workload-manager[.exe]` in the CWD | Explicit path to the `nvpair-workload-manager` binary the broker spawns for the cluster workload relay. Same optional semantics as `--node-info-path`: an absent default sibling means no workload relay (non-fatal); an invalid explicit path exits with an error |
| `--errors-path ` | `./nvpair-errors[.exe]` in the CWD | Explicit path to the `nvpair-errors` binary the broker spawns (with `--peer-sync`) for the service-error pipeline. Same optional semantics as `--node-info-path`: an absent default sibling means the error pipeline is disabled — producers' errors are dropped (non-fatal); an invalid explicit path exits with an error |
| `--engine-manager-path ` | `./nvpair-engine-manager[.exe]` in the CWD | Explicit path to the `nvpair-engine-manager` binary the broker spawns for engine management. Same optional semantics as `--node-info-path` |
@@ -91,43 +92,82 @@ Logs go to **stderr** (shared `applog` format, same as every other NVPAIR binary
On startup — **before** emitting `app:ready` — the broker spawns the scanner and (when available) node-info, `nvpair-proxy`, the workload-manager, and the cluster-manager as child processes over stdio. The proxy is spawned up front but doesn't gate `app:ready` — each of its facades announces its listen port asynchronously (see below). None of the auxiliary workers gate `app:ready`.
-**`nvpair-node-scanner`** (the consolidated discovery daemon) is spawned first. It pushes `discovery:node-discovered`, `discovery:node-updated`, and `discovery:node-removed` notifications into the broker, which maintains them in an in-memory map keyed by `id`. Clients query that map via `discovery:get-nodes` and — once they've opted in via `discovery:subscribe` — receive a `discovery:nodes-changed` notification on every store mutation. The raw `discovery:node-*` notifications are never forwarded as-is. The scanner polls healthy node-info endpoints on a staggered two-second cadence, backs consecutive remote failures off to a 30-second cap, and emits compact `discovery:node-telemetry` observations containing maximum GPU utilization, validity, and age; these remain internal to broker scheduling. The broker registers this node's local service ports (`ni`/`er`/`wl`/`cl`/`em`, plus `ol`/`lm` from the engine poller) with the daemon over the same link, so the daemon can advertise them all in one `_nvpair-node` record.
+**`nvpair-node-scanner`** (the consolidated discovery daemon) is spawned first. It pushes `discovery:node-discovered`, `discovery:node-updated`, and `discovery:node-removed` notifications into the broker, which maintains them in an in-memory map keyed by `id`. Clients query that map via `discovery:get-nodes` and — once they've opted in via `discovery:subscribe` — receive a `discovery:nodes-changed` notification on every store mutation. The raw `discovery:node-*` notifications are never forwarded as-is. The scanner polls healthy node-info endpoints on a staggered two-second cadence, backs consecutive remote failures off to a 30-second cap, and emits compact `discovery:node-telemetry` observations containing maximum GPU utilization, validity, and age; these remain internal to broker scheduling. The broker registers this node's local service ports (`ni`/`er`/`wl`/`cl`/`em`, plus `ol`/`lm`/`lc` from the engine poller) with the daemon over the same link, so the daemon can advertise them all in one `_nvpair-node` record.
**`nvpair-node-info`** is spawned next. It's a server, not an event source: it stands up the local `/v1/node-info` HTTP endpoint (GPU/CPU/memory inventory). It does not advertise itself — the broker registers its `ni` port with the scanner daemon, which carries it in the node record, and a peer's daemon fetches `/v1/node-info` over plain HTTP to enrich the node. The broker doesn't read anything back from node-info's stdout (drained and discarded). Spawning it is **optional**: if the binary can't be resolved (and no `--node-info-path` override was given) the broker logs a warning and continues serving discovery without it.
-**Engine advertising.** The broker runs an internal 5 s poll loop against local Ollama at its configured backend port and LM Studio (`GET /v1/models`) and reconciles this node's engine registration with the scanner daemon:
+**Engine advertising.** The broker runs an internal 5 s poll loop against local
+Ollama (`GET /`), LM Studio (`GET /v1/models`), and llama.cpp (`GET /health`)
+at their configured backend ports and reconciles this node's engine registration
+with the scanner daemon:
-- engine **up** → register `ol` / `lm` at the engine's real port, never the proxy's own, to prevent a self-forward loop;
+- engine **up** and facade **ready** → register `ol` / `lm` / `lc` at the
+ promoted facade port, never the private backend port;
- engine **down** → unregister it.
The daemon folds those registrations into this host's single `_nvpair-node` record, so a peer discovers the engine through the shared channel. The model list is not part of that registration — it's served over HTTP by `nvpair-engine-manager` (the `em` service, `GET /v1/models`) and enriched onto each node by the peer's daemon. There is no separate advertiser subprocess and no manual-advertise RPC.
**`nvpair-proxy`** is one process that fronts every enabled engine. It starts with no engine and no listener; the broker then sends it a `facade/enable` per engine, carrying that engine's port and any alias addresses. A flag could not express this, because the broker plans a different port for each engine. Each facade forwards inference to a node it discovers on the network and speaks its own engine's dialect.
-One process for all of them is deliberate. Between scheduler snapshots a facade takes short-lived reservations for work it has dispatched, and those live in the process — two processes each held half that picture, so simultaneous Ollama and LM Studio bursts could both pick the same node believing it idle. The cost is **shared fate**: a crash takes every facade down and the supervisor brings them all back together, reported once as `supervisor:subprocess-crashed:nvpair-proxy` rather than against one engine. Within the process the boundaries are finer — a facade that loses its bind race, cannot be moved to a free port, or panics while handling a request is withdrawn or answered with an error on its own, leaving the others serving.
+One process for all of them is deliberate. Between scheduler snapshots a facade
+takes short-lived reservations for work it has dispatched, and those live in
+the process; separate processes would each hold only part of that picture, so
+simultaneous bursts across engines could select the same node believing it idle.
+The cost is **shared fate**: a crash takes every facade down and the supervisor
+brings them all back together, reported once as
+`supervisor:subprocess-crashed:nvpair-proxy` rather than against one engine.
+Within the process the boundaries are finer — a facade that loses its bind
+race, cannot be moved to a free port, or panics while handling a request is
+withdrawn or answered with an error on its own, leaving the others serving.
Ollama's standalone default is `:11435`; with managed port ownership enabled (the default), the broker starts settings and engine-manager first, claims `:11434` with that facade, and only then moves a stopped default-port Ollama backend to a free port. Custom backend ports are preserved. When the inherited `OLLAMA_HOST` names a distinct local plaintext port, the broker also gives the facade that normalized loopback-only alias so clients already using the variable enter the same routing path; `localhost` reserves both canonical loopback families atomically, while remote and HTTPS targets are ignored. The alias port is reserved against every configured engine, local or remote engine start override, every facade's control plane, and the managed Ollama and LM Studio backend port plans, so a backend that has to move can never land on the alias. A running Ollama or unknown owner on either requested port is never stopped or moved: the primary uses a safe fallback when needed, and an occupied alias remains with its owner while the broker reports a warning.
+llama.cpp is a managed engine: engine-manager owns its lifecycle and starts it
+on its configured loopback port, `:8081` by default. The broker places its
+default-enabled facade on `:8080` or a safe fallback, without automatically
+relocating the engine or adding a compatibility-port ownership gate. Fallback
+selection excludes the default engine port and other engines' reserved ports.
+Explicit engine and proxy settings are preserved; a bind failure on an explicitly
+chosen proxy port is reported instead of selecting a fallback.
+
For automatic model-bearing inference, every facade combines scheduler pending counts and GPU pressure with the process-wide reservation map under one lock before forwarding, so concurrent requests distribute without an artificial delay or a round trip through the scheduler. A reservation is released when its request ends and moves with a failover, so a node stops counting as loaded as soon as it stops working. Manual pins, model-owner tiers, and the complete failover list keep their existing precedence. The broker otherwise treats the proxy as **optional and non-fatal**.
Each facade announces its bound port **asynchronously**, via an engine-addressed `ready` notification emitted once its HTTP listener is up. The broker records readiness per engine and exposes it through that engine's `-proxy:get-status` request — per engine, because the facades bind different ports and a single port for the process would be whichever readied last. Because `ready` arrives after `app:ready` (and the proxy is optional), clients learn a port by **polling** `-proxy:get-status` rather than assuming it from `app:ready`.
-The proxy is a full bidirectional JSON-RPC peer with a control plane (node selection, manual nodes, ...) and an event stream. The broker acts as a **generic relay** in both directions: any request a client sends under the `ollama-proxy:` namespace (other than the reserved broker-local ones) is forwarded to that engine's facade and the response relayed straight back (see `ollama-proxy:` below), and every notification the facade emits is re-emitted to subscribed clients as `ollama-proxy:` (see `ollama-proxy:` below).
+The proxy is a full bidirectional JSON-RPC peer with a control plane (node
+selection, manual nodes, ...) and an event stream. The broker acts as a
+**generic relay** in both directions: any request a client sends under an
+enabled `-proxy:` namespace (other than reserved broker-local ones) is
+forwarded to that engine's facade and the response relayed straight back, and
+every facade notification is re-emitted to subscribed clients under the same
+component namespace.
Client namespaces are unchanged by the process collapse, but the wire inside is not. Because one process holds every facade, a message on that link carries the engine it concerns: the client's `ollama-proxy:` prefix comes off and a bare `ollama:` facade address goes on. The two are not interchangeable — `ollama-proxy:` is how a client addresses the component, `ollama:` is how a message addresses a facade inside the process — so the relay is a translation between them rather than a strip. Process-scoped methods (`log/set-level`, the scheduler's `node/set-priority`) carry no address, and `facade/enable` names its engine in the payload because it runs before that facade exists.
-Two classes of proxy notification are **not** re-emitted under the `ollama-proxy:` namespace, because neither is a proxy control-plane event:
+Two classes of proxy notification are **not** re-emitted under any
+`-proxy:` namespace, because neither is a proxy control-plane event:
- The `workload:*` lifecycle events the proxy fires per inference request. Those are workload-manager traffic — see the workload-manager paragraph below for how they're routed.
- `node/activity`, which a proxy raises while a peer's engine is streaming response bytes back through it. That is discovery input: the broker forwards it to `nvpair-node-scanner` as a `discovery:node-activity` notification, where it counts as proof the peer is alive and cancels the eviction it would otherwise face for failing a liveness probe it had no spare CPU to answer. No client has any use for a per-request liveness frame. The handoff is a bounded queue drained by one goroutine — reports arrive for as long as inference streams, so a wedged scanner must not be able to stall the proxy reader, and a dropped report only means the scanner falls back to probing a node that will very likely answer.
**`nvpair-workload-manager`** is the cluster workload relay, and it's the only worker the broker talks to **bidirectionally over a notification-only link** (no id-bearing request/response). It supervises it the same optional, non-fatal way as node-info / proxy: a missing default sibling (and no `--workload-manager-path`) just means no cluster workload relay. The broker plays the workload **broker** role between the proxy and the manager:
-- **Outbound (proxy -> broker -> manager -> peers).** When either proxy emits a `workload:started` / `workload:completed` / `workload:errored`, the broker stamps the local stable `hostUuid` onto `params.workloadInfo.originatedFrom`, applies the transition to its authoritative workload store, and fans the accepted update to the scheduler before forwarding the original lifecycle frame to the manager. The manager broadcasts it to peer nodes. With no manager supervised the event still updates local scheduling and subscribed clients, but is not broadcast.
+- **Outbound (proxy -> broker -> manager -> peers).** When a facade emits a `workload:started` / `workload:completed` / `workload:errored`, the broker stamps the local stable `hostUuid` onto `params.workloadInfo.originatedFrom`, applies the transition to its authoritative workload store, and fans the accepted update to the scheduler before forwarding the original lifecycle frame to the manager. The manager broadcasts it to peer nodes. With no manager supervised the event still updates local scheduling and subscribed clients, but is not broadcast.
- **Inbound (peers -> manager -> broker).** The manager translates peer-origin lifecycle events into `workloads:upsert` and peer-origin removals into `workloads:remove` on stdout. The broker applies each accepted transition to the same store, fans it to the scheduler, and relays it to clients subscribed via `workloads:subscribe`.
- **Local echo.** Local-origin proxy workloads are also emitted to the same `workloads:*` client stream (lifecycle translated to `workloads:upsert`), so a subscribed client sees a coherent cluster-wide view — its own workloads alongside peers'.
-**`nvpair-job-scheduler`** consumes the accepted workload stream, compact GPU telemetry, and discovery snapshot. It smooths fresh utilization into pressure 0–3, uses neutral pressure 1 for invalid/missing/older-than-10-second samples, and orders by `pending + gpuPressure`, then pressure, then stable UUID. Load is node-wide across Ollama and LM Studio because both normally contend for the same resources. Each engine-specific `schedule:priority` carries `{engine,nodes,ranks}` and refreshes when order, pending counts, or pressure changes. The broker caches, generation-orders, and replays the full `{nodes,ranks}` snapshot to the matching proxy, where a newly delivered snapshot resets optimistic reservation deltas. On scheduler spawn/restart the broker replays active workloads and telemetry before discovery, then resumes all three live feeds.
+**`nvpair-job-scheduler`** consumes the accepted workload stream, compact GPU
+telemetry, and discovery snapshot. It smooths fresh utilization into pressure
+0–3, uses neutral pressure 1 for invalid/missing/older-than-10-second samples,
+and orders by `pending + gpuPressure`, then pressure, then stable UUID. Load is
+node-wide across Ollama, LM Studio, and llama.cpp because they normally contend
+for the same resources. Each engine-specific `schedule:priority` carries
+`{engine,nodes,ranks}` and refreshes when order, pending counts, or pressure
+changes. The broker caches, generation-orders, and replays the full
+`{nodes,ranks}` snapshot to the matching facade, where a newly delivered
+snapshot resets optimistic reservation deltas. On scheduler spawn/restart the
+broker replays active workloads and telemetry before discovery, then resumes all
+three live feeds.
`schedule:priority` and `node/set-priority` are internal worker contracts: the broker does not expose either notification to its connected client.
@@ -205,7 +245,11 @@ Two classes of proxy notification are **not** re-emitted under the `ollama-proxy
**Opt-in.** Only delivered to a peer that has called `workloads:subscribe`; silent otherwise. Once subscribed, the broker pushes a `workloads:upsert` whenever a workload is created or its state changes, and a `workloads:remove` when one is retired. The stream is the union of two sources, in the same shape regardless of origin:
-- **Local workloads** — the `workload:*` lifecycle events the supervised Ollama and LM Studio proxies emit per inference request, stamped with this host's stable `hostUuid` (`originatedFrom`) and translated to `workloads:upsert`. The proxy also fills in `scheduledOn` with the destination node's `hostUuid`; the broker passes that through unchanged.
+- **Local workloads** — the `workload:*` lifecycle events the enabled proxy
+ facades emit per inference request, stamped with this host's stable `hostUuid`
+ (`originatedFrom`) and translated to `workloads:upsert`. The proxy also fills
+ in `scheduledOn` with the destination node's `hostUuid`; the broker passes
+ that through unchanged.
- **Peer workloads** — the `workloads:upsert` / `workloads:remove` the `nvpair-workload-manager` relays from other nodes after validating and de-duplicating their broadcasts.
- **Inferred workloads** — a `workloads:upsert` transitioning a workload to `failed` that **no origin ever sent**. The broker synthesizes one in two situations: when a node leaves discovery while workloads are pinned to it, and when a remote origin that is still present stops re-asserting a workload this node believes is running (the origin's re-sync heartbeat asserts each of its active workloads indefinitely, so prolonged silence about one means it is finished or the origin is gone). Both are recorded as *inferred*, so the origin's next authoritative event overrides them; a client should treat a `failed` as the broker's best current answer rather than proof the origin reported a failure, and its `error` text names the reason. Workloads this node originated or is itself executing are never inferred about.
@@ -383,21 +427,24 @@ Two relay-specific error cases:
- If no proxy is being supervised (or it has exited), the broker replies with error `-32000` `"ollama-proxy not available"`.
- `ollama-proxy:shutdown` is **refused** with error `-32601` — the broker owns the proxy's lifecycle, so a client can't terminate it independently. Shut the broker down instead (which tears the proxy down with it).
-#### `ollama-proxy:set-port`
+#### `ollama-proxy:set-port` / `lmstudio-proxy:set-port` / `llamacpp-proxy:set-port`
-**Intercepted, not relayed verbatim.** A port-only caller — `nvpair-tui` is the one in tree — gets to move a single port without rendering the whole launch settings form, but the change still runs through the same authoritative settings operation the desktop editor uses, so a port set from the terminal cannot diverge from one set from the UI. The broker reads the engine's current settings, substitutes the requested proxy port, and applies the result.
+**Intercepted, not relayed verbatim.** For every engine profile, a port-only caller — `nvpair-tui` is the one in tree — gets to move a single port without rendering the whole launch settings form, but the change still runs through the same authoritative settings operation the desktop editor uses, so a port set from the terminal cannot diverge from one set from the UI. The namespace selects the engine. The broker reads that engine's current settings, substitutes the requested proxy port, and applies the result while preserving the server port and launch arguments. Accepted settings are saved in the settings journal.
-A **requested port that is already in use is refused** with error `-32000 "port %d is already in use"`. The broker does not pick a different port on the caller's behalf: silently binding somewhere else left clients pointed at a port nothing was listening on. Retry with a free port. A request that collides with an inherited `OLLAMA_HOST` alias is refused with its own message naming that alias. The response echoes the requested port (`{"port": }`) once it is bound.
+A malformed request, missing port, or port outside `1`–`65535` is refused with error `-32602 "port must be between 1 and 65535"`. A **requested port that is already in use or reserved** by a PAIR service, another configured engine/proxy, or the engine's own server port is refused with error `-32000 "resolve settings errors and port conflicts before applying"`. The broker does not pick a different port on the caller's behalf. Retry with a free port. A request that collides with an inherited `OLLAMA_HOST` alias is refused with its own message naming that alias. The response echoes the requested port (`{"port": }`) once it is bound.
```json
{"jsonrpc":"2.0","id":9,"method":"ollama-proxy:set-port","params":{"port":11500}}
+{"jsonrpc":"2.0","id":10,"method":"llamacpp-proxy:set-port","params":{"port":8082}}
```
-Automatic conflict resolution still exists, but only for a port the user did not just choose: when the proxy announces a (re)bound port on startup and a running engine has since taken it, the broker steers the proxy to a free port and surfaces a sticky `warning` into the errors pipeline (id `ollama-proxy:port-bumped`, `action:"none"`) explaining the move. That path **never changes an engine's port** — only the proxy is moved. Error `-32000 "ollama-proxy not available"` when no proxy is supervised.
+Automatic conflict resolution still exists, but only for a port the user did not just choose: when the Ollama proxy announces a (re)bound port on startup and a running engine has since taken it, the broker steers the proxy to a free port and surfaces a sticky `warning` into the errors pipeline (id `ollama-proxy:port-bumped`, `action:"none"`) explaining the move. That path **never changes an engine's port** — only the proxy is moved. Port setters also return `-32000` if the settings operation cannot run, including when the engine manager or proxy is unavailable.
#### `lmstudio-proxy:get-status` / `lmstudio-proxy:subscribe` / `lmstudio-proxy:unsubscribe` / `lmstudio-proxy:` (generic relay)
-The LM Studio counterpart of the `ollama-proxy:*` surface runs the supervised `lmstudio-proxy` on compatibility port `:1234` and tracks the managed LM Studio backend on `:1235`. With managed port ownership enabled (the default), the broker identifies and moves an existing LM Studio server through engine-manager before allowing the proxy to claim `1234`; unknown owners are left untouched and force a warned proxy fallback. Disabling managed ownership preserves explicit custom backend and proxy ports. `lmstudio-proxy:get-status` reports the actual bound port; `lmstudio-proxy:subscribe` / `lmstudio-proxy:unsubscribe` opt into / out of its `lmstudio-proxy:` stream; and any other `lmstudio-proxy:` is relayed verbatim with the prefix stripped (`nodes/list`, `node/select`, `node/add-manual`, `node/remove-manual`, ...). `lmstudio-proxy:shutdown` is refused because the broker owns lifecycle ordering. Workload and error events feed the shared streams exactly as Ollama's do.
+The LM Studio counterpart of the `ollama-proxy:*` surface runs the supervised `lmstudio-proxy` on compatibility port `:1234` and tracks the managed LM Studio backend on `:1235`. With managed port ownership enabled (the default), the broker identifies and moves an existing LM Studio server through engine-manager before allowing the proxy to claim `1234`; unknown owners are left untouched and force a warned proxy fallback. Disabling managed ownership preserves explicit custom backend and proxy ports. `lmstudio-proxy:get-status` reports the actual bound port; `lmstudio-proxy:subscribe` / `lmstudio-proxy:unsubscribe` opt into / out of its `lmstudio-proxy:` stream; `lmstudio-proxy:set-port` uses the settings operation described above; and other `lmstudio-proxy:` requests are relayed with the prefix translated (`nodes/list`, `node/select`, `node/add-manual`, `node/remove-manual`, ...). `lmstudio-proxy:shutdown` is refused because the broker owns lifecycle ordering. Workload and error events feed the shared streams exactly as Ollama's do.
+
+The same broker-local status, subscription, and port-setting methods apply to `llamacpp-proxy:*`. llama.cpp port changes use the same validation, journal, and application path; remaining methods use the generic facade relay, with `llamacpp-proxy:shutdown` refused.
#### `ollama-proxy:subscribe`
@@ -483,7 +530,7 @@ Opt into / out of the `engine:` stream (off by default). Acks `{ subscrib
Any other `engine:*` request is forwarded to `nvpair-engine-manager` verbatim and its response relayed straight back. This covers the whole engine control plane: `engine:get-installed`, `engine:describe`, `engine:status`, `engine:install`, `engine:uninstall`, `engine:start`, `engine:stop`, `engine:restart`, `engine:action`, `engine:logs`, `engine:errors`, `engine:models`. Lifecycle ops run for minutes (reporting progress via the `engine:install-progress` / `engine:state-changed` push events), so the relay imposes **no broker-side timeout** — fire the request and watch the event stream for the outcome. Error `-32000 "engine-manager not available"` when no engine-manager is supervised.
-`engine:set-port` is **not** in that generic set. Like `proxy:set-port` it is intercepted and run through the authoritative settings operation, so moving an engine's server port from a port-only caller validates and restarts exactly as the full editor does, and persists as a manifest override that survives a restart. Its response is the engine's `engine:status` result.
+`engine:set-port` is **not** in that generic set. Like `-proxy:set-port` it is intercepted and run through the authoritative settings operation, so moving an engine's server port from a port-only caller validates and restarts exactly as the full editor does, and persists as a manifest override that survives a restart. Its response is the engine's `engine:status` result.
#### `settings/` (generic relay)
@@ -569,7 +616,11 @@ Attach to a pre-existing endpoint:
## What this version intentionally does NOT do (yet)
-- **Engine-advertise control surface.** Engine registration is auto-driven only: the broker tracks local ollama / LM Studio on their fixed coordinates and registers `ol` / `lm` with the daemon while up. There's no manual-advertise RPC (custom service, port, name, or TXT), and no way to advertise anything other than the detected engines.
+- **Engine-advertise control surface.** Engine registration is auto-driven only:
+ the broker tracks local Ollama, LM Studio, and llama.cpp and registers `ol`,
+ `lm`, or `lc` with the daemon while each is up. There's no manual-advertise
+ RPC (custom service, port, name, or TXT), and no way to advertise anything
+ other than the detected engines.
- **node-info control surface.** node-info is spawned and torn down with the broker, and the broker pushes it only two things over stdin: the log level, and this node's cluster principal (`nodeinfo:set-cluster-identity`, sent on spawn and on every membership or pin-set change, because node-info holds no cluster dir and so cannot read membership itself). Otherwise it's hands-off: the broker registers its port with the daemon (which enriches over plain HTTP) but doesn't pass through TLS material (`--cert` / `--key` / `--client-ca`) or a custom `--port`, and exposes no RPC to query or reconfigure it. It runs with its own defaults plus those two pushes.
- **Manual-node persistence across restarts.** `nvpair-manual-nodes` keeps its entries only in memory and the broker holds no authoritative copy, so a manual-nodes crash-and-restart loses the user's manual nodes (the broker evicts the orphaned entries from the snapshot; clients must re-add them).
- **Per-event push semantics.** `discovery:nodes-changed` always carries the full current snapshot, not a delta. For small N this is fine and lets the client treat the payload as authoritative without state reconciliation. `errors:update` is likewise a full snapshot.
diff --git a/services/nvpair-ui-broker/advertiser.go b/services/nvpair-ui-broker/advertiser.go
index d48097e7..11e864ba 100644
--- a/services/nvpair-ui-broker/advertiser.go
+++ b/services/nvpair-ui-broker/advertiser.go
@@ -188,6 +188,43 @@ func (b *Broker) reconcileAdvertiseLMStudio(client *http.Client) {
}
}
+// runAutoAdvertiseEngine reconciles an engine using the configured port
+// recorded in its runtime profile, without compatibility-port reconciliation.
+func (b *Broker) runAutoAdvertiseEngine(ctx context.Context, profile engineProxyProfile) {
+ client := &http.Client{Timeout: 2 * time.Second}
+ ticker := time.NewTicker(autoAdvertiseInterval)
+ defer ticker.Stop()
+
+ b.reconcileAdvertiseEngine(profile, client)
+ for {
+ select {
+ case <-ctx.Done():
+ return
+ case <-ticker.C:
+ b.reconcileAdvertiseEngine(profile, client)
+ }
+ }
+}
+
+func (b *Broker) reconcileAdvertiseEngine(profile engineProxyProfile, client *http.Client) {
+ b.engineConfigMu.Lock()
+ defer b.engineConfigMu.Unlock()
+
+ enginePort := int(b.engineProxy(profile).backendPort.Load())
+ proxyPort := b.engineProxyListenPort(profile)
+ up := enginePort > 0 &&
+ proxyPort > 0 &&
+ enginePort != proxyPort &&
+ checkEngineHealth(profile, client, enginePort)
+ if up {
+ b.registerService(noderec.RegisterParams{Service: profile.DiscoveryService, Port: proxyPort})
+ b.setProxyLocalBackend(b.engineProxyHandle(profile), profile.Name, enginePort, true)
+ return
+ }
+ b.unregisterService(profile.DiscoveryService)
+ b.setProxyLocalBackend(b.engineProxyHandle(profile), profile.Name, enginePort, false)
+}
+
// proxyLocalBackend is the node/set-local-backend payload: the loopback engine
// the proxy's cluster mTLS ingress forwards to, and the proxy's own self
// candidate on the local routing path.
diff --git a/services/nvpair-ui-broker/advertiser_test.go b/services/nvpair-ui-broker/advertiser_test.go
index a32c2cbe..b2435ce6 100644
--- a/services/nvpair-ui-broker/advertiser_test.go
+++ b/services/nvpair-ui-broker/advertiser_test.go
@@ -6,9 +6,13 @@ package main
import (
"encoding/json"
"net"
+ "net/http"
+ "net/http/httptest"
+ "strconv"
"testing"
"time"
+ "nvpair-shared/noderec"
"nvpair-ui-broker/relay"
)
@@ -107,6 +111,108 @@ func TestLMStudioFallbackDoesNotOverwriteKnownBackend(t *testing.T) {
}
}
+func TestEngineAdvertiserTracksEngineHealth(t *testing.T) {
+ backend := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
+ w.WriteHeader(http.StatusOK)
+ }))
+ defer backend.Close()
+ _, portText, err := net.SplitHostPort(backend.Listener.Addr().String())
+ if err != nil {
+ t.Fatal(err)
+ }
+ backendPort, err := strconv.Atoi(portText)
+ if err != nil {
+ t.Fatal(err)
+ }
+ proxyPort := 44000
+ if backendPort == proxyPort {
+ proxyPort++
+ }
+
+ profile := testDefaultEngineProxyProfile()
+ profile.DiscoveryService = noderec.ServiceLMStudio
+ b := brokerWithEngineProxyProfile(profile)
+ b.regCache = relay.NewRegistrationCache()
+ b.engineProxy(profile).backendPort.Store(int32(backendPort))
+ updates := attachAdvertiserProxy(t, b, profile, proxyPort)
+
+ b.reconcileAdvertiseEngine(profile, backend.Client())
+ registrations := b.regCache.Snapshot()
+ if len(registrations) != 1 || registrations[0].Service != profile.DiscoveryService ||
+ registrations[0].Port != proxyPort {
+ t.Fatalf("healthy registration = %+v, want %s on %d", registrations, profile.DiscoveryService, proxyPort)
+ }
+ if got := <-updates; !got.Healthy || got.Port != backendPort || got.Engine != profile.Name {
+ t.Fatalf("healthy local backend = %+v", got)
+ }
+
+ backend.Close()
+ b.reconcileAdvertiseEngine(profile, backend.Client())
+ if got := b.regCache.Snapshot(); len(got) != 0 {
+ t.Fatalf("unhealthy engine remained advertised: %+v", got)
+ }
+ if got := <-updates; got.Healthy || got.Port != backendPort {
+ t.Fatalf("unhealthy local backend = %+v", got)
+ }
+}
+
+func TestEngineAdvertiserRejectsSelfForwardLoop(t *testing.T) {
+ profile := testDefaultEngineProxyProfile()
+ profile.DiscoveryService = noderec.ServiceLMStudio
+ b := brokerWithEngineProxyProfile(profile)
+ b.regCache = relay.NewRegistrationCache()
+ b.regCache.Register(noderec.RegisterParams{Service: profile.DiscoveryService, Port: 44000})
+ b.engineProxy(profile).backendPort.Store(44000)
+ updates := attachAdvertiserProxy(t, b, profile, 44000)
+
+ // A nil client proves the collision check short-circuits before probing the
+ // facade as though it were the backend.
+ b.reconcileAdvertiseEngine(profile, nil)
+ if got := b.regCache.Snapshot(); len(got) != 0 {
+ t.Fatalf("self-forwarding facade remained advertised: %+v", got)
+ }
+ if got := <-updates; got.Healthy {
+ t.Fatalf("self-forwarding backend remained healthy: %+v", got)
+ }
+}
+
+func attachAdvertiserProxy(
+ t *testing.T,
+ b *Broker,
+ profile engineProxyProfile,
+ port int,
+) <-chan proxyLocalBackend {
+ t.Helper()
+ proxyClient, proxyServer := net.Pipe()
+ t.Cleanup(func() {
+ _ = proxyClient.Close()
+ _ = proxyServer.Close()
+ })
+ proxy := &proxyProcess{
+ peer: NewPeer(NewCodec(proxyClient)),
+ facadeState: readyFacade(profile.Name, port),
+ }
+ go proxy.peer.Serve(nil, nil)
+ b.setEngineProxyHandle(profile, proxy)
+
+ updates := make(chan proxyLocalBackend, 4)
+ go func() {
+ codec := NewCodec(proxyServer)
+ for {
+ msg, err := codec.Read()
+ if err != nil {
+ return
+ }
+ var update proxyLocalBackend
+ if json.Unmarshal(msg.Params, &update) == nil {
+ updates <- update
+ }
+ _ = codec.Respond(msg.ID, map[string]bool{"ok": true})
+ }
+ }()
+ return updates
+}
+
// TestProxyListenPortNoProxy: with no proxy supervised, proxyListenPort is 0,
// so the self-forward collision check (port == proxy port) never falsely trips.
func TestProxyListenPortNoProxy(t *testing.T) {
diff --git a/services/nvpair-ui-broker/broker.go b/services/nvpair-ui-broker/broker.go
index c5da7fd1..4e1dcacb 100644
--- a/services/nvpair-ui-broker/broker.go
+++ b/services/nvpair-ui-broker/broker.go
@@ -133,11 +133,11 @@ type SubscriptionResult struct {
Subscribed bool `json:"subscribed"`
}
-// ProxyStatusResult is the response to "ollama-proxy:get-status". Ready is false
-// (and Port 0) until the supervised ollama-proxy has emitted its "ready"
-// notification — or always, if no proxy is being supervised. Clients poll
-// this to learn where the local proxy is listening, since the proxy is
-// optional and comes up asynchronously after app:ready.
+// ProxyStatusResult is the response to "-proxy:get-status". Ready is
+// false (and Port 0) until that facade has emitted its "ready" notification —
+// or always, if no proxy is being supervised. Clients poll this to learn where
+// the local facade is listening, since the proxy is optional and comes up
+// asynchronously after app:ready.
type ProxyStatusResult struct {
Ready bool `json:"ready"`
Port int `json:"port"`
@@ -537,14 +537,18 @@ func (b *Broker) restoreEnabledEnginesAfterPortGate(ctx context.Context) bool {
func (b *Broker) runEngineAvailabilityAfterPortGates(
ctx context.Context,
- runOllama func(context.Context),
- runLMStudio func(context.Context),
+ runners ...func(context.Context),
) bool {
if !b.restoreEnabledEnginesAfterPortGate(ctx) {
return false
}
- go runOllama(ctx)
- runLMStudio(ctx)
+ for index, run := range runners {
+ if index == len(runners)-1 {
+ run(ctx)
+ break
+ }
+ go run(ctx)
+ }
return true
}
@@ -754,6 +758,16 @@ func (b *Broker) proxyBringUpContext() (context.Context, context.CancelFunc) {
// an inherited host variable, so the others ignore it.
func (b *Broker) enableEngineFacade(
ctx context.Context, pp *proxyProcess, profile engineProxyProfile, alias ollamaHostAlias,
+) error {
+ return b.enableEngineFacadeWithPortCheck(ctx, pp, profile, alias, tcpPortAvailable)
+}
+
+func (b *Broker) enableEngineFacadeWithPortCheck(
+ ctx context.Context,
+ pp *proxyProcess,
+ profile engineProxyProfile,
+ alias ollamaHostAlias,
+ available func(int) bool,
) error {
switch profile.Name {
case ollamaProxyProfile.Name:
@@ -761,7 +775,14 @@ func (b *Broker) enableEngineFacade(
case lmstudioProxyProfile.Name:
return b.enableProxyFacadeWithFallback(ctx, pp, b.lmstudioFacadeSpec(), b.lmstudioFallbackPort)
default:
- return fmt.Errorf("no facade spec for engine %q", profile.Name)
+ return b.enableProxyFacadeWithFallback(
+ ctx,
+ pp,
+ b.defaultEngineFacadeSpec(profile),
+ func(failed int) int {
+ return b.defaultEngineFallbackPortWithCheck(profile, failed, available)
+ },
+ )
}
}
@@ -787,7 +808,7 @@ func (b *Broker) blockAndFinishEngineProxy(profile engineProxyProfile) {
}
b.finishLMStudioProxyTerminal()
default:
- slog.Warn("no terminal handling for engine", "engine", profile.Name)
+ // Other engines have no compatibility-port claim or startup gate.
}
}
@@ -981,8 +1002,13 @@ func (b *Broker) forwardProxyProcessNotification(
slog.Debug("ignoring unaddressed proxy notification", "method", bare)
}
default:
- slog.Warn("proxy addressed a notification to an unknown engine",
- "engine", engine, "method", bare)
+ profile, known := engineProxyProfileFor(engine)
+ if known {
+ b.forwardDefaultEngineProxyNotification(profile, method, params)
+ return
+ }
+ slog.Warn("proxy addressed a notification without a handler",
+ "engine", engine, "method", bare, "known", known)
}
}
@@ -2196,11 +2222,24 @@ func (b *Broker) Serve(ctx context.Context) error {
}
}
- // Restore engines and begin both advertising loops only after both proxy
+ // Restore engines and begin advertising only after both managed proxy
// startup attempts have established either readiness or a terminal outcome.
// This prevents a restored engine from taking a persisted proxy port before
// the broker can resolve ownership.
- go b.runEngineAvailabilityAfterPortGates(ctx, b.runAutoAdvertise, b.runAutoAdvertiseLMStudio)
+ availabilityRunners := []func(context.Context){b.runAutoAdvertise, b.runAutoAdvertiseLMStudio}
+ for _, profile := range engineProxyProfiles {
+ if !b.proxyEnabled(profile) {
+ continue
+ }
+ if profile.Name == ollamaProxyProfile.Name || profile.Name == lmstudioProxyProfile.Name {
+ continue
+ }
+ profile := profile
+ availabilityRunners = append(availabilityRunners, func(ctx context.Context) {
+ b.runAutoAdvertiseEngine(ctx, profile)
+ })
+ }
+ go b.runEngineAvailabilityAfterPortGates(ctx, availabilityRunners...)
// nvpair-workload-manager is another auxiliary worker: it relays local
// workload lifecycle events to peer nodes and surfaces peer events
@@ -3142,6 +3181,13 @@ func (b *Broker) handleMessage(msg *Message) {
return
}
+ if profile, ok := engineProxyProfileForMethod(msg.Method); ok {
+ method := strings.TrimPrefix(msg.Method, profile.ComponentName()+":")
+ if b.handleEngineProxyBrokerRequest(profile, method, msg) {
+ return
+ }
+ }
+
switch msg.Method {
case "engine:get-settings", "engine:preview-settings", "engine:apply-settings":
go b.handleEngineSettings(msg)
@@ -3189,94 +3235,8 @@ func (b *Broker) handleMessage(msg *Message) {
log.Printf("failed to respond to discovery:unsubscribe: %v", err)
}
- case "ollama-proxy:get-status":
- // Answered locally from the proxy handle's captured state — no
- // round-trip to the proxy. When no proxy is supervised the
- // zero value ({ready:false, port:0}) is a valid "not available"
- // answer, so the method never errors.
- var result ProxyStatusResult
- if p := b.getProxy(); p != nil {
- ready, port := p.Status(ollamaProxyProfile.Name)
- result.Ready = ready
- result.Port = port
- }
- if err := b.codec.Respond(msg.ID, result); err != nil {
- log.Printf("failed to respond to proxy:get-status: %v", err)
- }
-
- case "ollama-proxy:subscribe":
- b.proxyMu.Lock()
- wasSubscribed := b.setEngineProxySubscribed(ollamaProxyProfile, true)
- b.proxyMu.Unlock()
- if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: true}); err != nil {
- log.Printf("failed to respond to proxy:subscribe: %v", err)
- }
- // On a fresh subscription replay the proxy's last "ready" payload
- // (if it has come up) as a baseline proxy:ready, so a subscriber
- // learns the port without a separate proxy:get-status. Sent after
- // the ack. A redundant re-subscribe doesn't re-emit.
- if !wasSubscribed {
- if p := b.getProxy(); p != nil {
- if rp := p.ReadyParams(ollamaProxyProfile.Name); rp != nil {
- if err := b.codec.Notify("ollama-proxy:ready", rp); err != nil {
- slog.Warn("emit baseline proxy:ready failed", "err", err)
- }
- }
- }
- }
-
- case "ollama-proxy:unsubscribe":
- b.proxyMu.Lock()
- b.setEngineProxySubscribed(ollamaProxyProfile, false)
- b.proxyMu.Unlock()
- if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: false}); err != nil {
- log.Printf("failed to respond to proxy:unsubscribe: %v", err)
- }
-
case "engine:set-port":
go b.handleSettingsPortRPC(msg, "")
- case "ollama-proxy:set-port":
- go b.handleSettingsPortRPC(msg, "ollama")
- case "lmstudio-proxy:set-port":
- go b.handleSettingsPortRPC(msg, "lmstudio")
-
- case "lmstudio-proxy:get-status":
- // Answered locally from the lmstudio-proxy handle's captured state,
- // mirroring proxy:get-status. Zero value when none is supervised.
- var result ProxyStatusResult
- if p := b.getLMStudioProxy(); p != nil {
- ready, port := p.Status(lmstudioProxyProfile.Name)
- result.Ready = ready
- result.Port = port
- }
- if err := b.codec.Respond(msg.ID, result); err != nil {
- log.Printf("failed to respond to lmstudio-proxy:get-status: %v", err)
- }
-
- case "lmstudio-proxy:subscribe":
- b.proxyMu.Lock()
- wasSubscribed := b.setEngineProxySubscribed(lmstudioProxyProfile, true)
- b.proxyMu.Unlock()
- if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: true}); err != nil {
- log.Printf("failed to respond to lmstudio-proxy:subscribe: %v", err)
- }
- if !wasSubscribed {
- if p := b.getLMStudioProxy(); p != nil {
- if rp := p.ReadyParams(lmstudioProxyProfile.Name); rp != nil {
- if err := b.codec.Notify("lmstudio-proxy:ready", rp); err != nil {
- slog.Warn("emit baseline lmstudio-proxy:ready failed", "err", err)
- }
- }
- }
- }
-
- case "lmstudio-proxy:unsubscribe":
- b.proxyMu.Lock()
- b.setEngineProxySubscribed(lmstudioProxyProfile, false)
- b.proxyMu.Unlock()
- if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: false}); err != nil {
- log.Printf("failed to respond to lmstudio-proxy:unsubscribe: %v", err)
- }
case "workloads:subscribe":
b.workloadsMu.Lock()
@@ -3364,8 +3324,8 @@ func (b *Broker) handleMessage(msg *Message) {
default:
// Any remaining method under an engine's : prefix is
// relayed verbatim to that engine's proxy (the reserved broker-local
- // ones — get-status and the subscription methods — are handled by
- // their own cases above). This makes the broker a thin pass-through
+ // ones — get-status, set-port and the subscription methods — are handled by
+ // the profile-driven block above). This makes the broker a thin pass-through
// for each proxy's whole control plane without enumerating methods.
//
// The prefixes are the engines' ComponentName values, so this loop
diff --git a/services/nvpair-ui-broker/broker_lifecycle_test.go b/services/nvpair-ui-broker/broker_lifecycle_test.go
index f60fcbe5..ef17b83b 100644
--- a/services/nvpair-ui-broker/broker_lifecycle_test.go
+++ b/services/nvpair-ui-broker/broker_lifecycle_test.go
@@ -54,7 +54,7 @@ func TestEngineAvailabilityWaitsForBothProxyOutcomes(t *testing.T) {
restore <- msg.Method
}
}()
- advertised := make(chan string, 2)
+ advertised := make(chan string, 3)
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
done := make(chan bool, 1)
@@ -63,6 +63,7 @@ func TestEngineAvailabilityWaitsForBothProxyOutcomes(t *testing.T) {
ctx,
func(context.Context) { advertised <- "ollama" },
func(context.Context) { advertised <- "lmstudio" },
+ func(context.Context) { advertised <- "llamacpp" },
)
}()
@@ -92,12 +93,12 @@ func TestEngineAvailabilityWaitsForBothProxyOutcomes(t *testing.T) {
t.Fatal("enabled-engine restore did not run after both proxy outcomes")
}
seen := map[string]bool{}
- for len(seen) < 2 {
+ for len(seen) < 3 {
select {
case got := <-advertised:
seen[got] = true
case <-time.After(2 * time.Second):
- t.Fatalf("advertising did not start for both engines: %v", seen)
+ t.Fatalf("advertising did not start for every engine: %v", seen)
}
}
if !<-done {
diff --git a/services/nvpair-ui-broker/enginefacade_test.go b/services/nvpair-ui-broker/enginefacade_test.go
new file mode 100644
index 00000000..9ccb01b8
--- /dev/null
+++ b/services/nvpair-ui-broker/enginefacade_test.go
@@ -0,0 +1,251 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package main
+
+import (
+ "context"
+ "encoding/json"
+ "net"
+ "sync/atomic"
+ "testing"
+ "time"
+
+ "nvpair-shared/engines"
+ settings "nvpair-shared/enginesettings"
+)
+
+func testDefaultEngineProxyProfile() engineProxyProfile {
+ return engineProxyProfile{
+ Engine: engines.Engine{
+ Name: "fixedtest",
+ DisplayName: "Fixed Test",
+ FacadePort: 1233,
+ EnginePortBase: 1234,
+ PortFile: "fixedtest-proxy-port.json",
+ },
+ Ownership: managedEngine,
+ HealthProbePath: "/health",
+ }
+}
+
+func brokerWithEngineProxyProfile(profile engineProxyProfile) *Broker {
+ b := &Broker{}
+ b.engineProxiesOnce.Do(func() {
+ b.engineProxies = map[string]*engineProxyRuntime{
+ profile.Name: {profile: profile},
+ }
+ })
+ return b
+}
+
+func TestDefaultEngineFacadeRetriesAwayFromReservedPorts(t *testing.T) {
+ isolateOllamaHostTestConfig(t)
+ profile := testDefaultEngineProxyProfile()
+ b := brokerWithEngineProxyProfile(profile)
+
+ proxyClient, proxyServer := net.Pipe()
+ t.Cleanup(func() {
+ _ = proxyClient.Close()
+ _ = proxyServer.Close()
+ })
+ proxy := &proxyProcess{peer: NewPeer(NewCodec(proxyClient))}
+ go proxy.peer.Serve(nil, nil)
+ attempts := make(chan enableFacadeRequest, 2)
+ serveFacadeEnable(t, proxyServer, map[int]bool{profile.FacadePort: true}, attempts)
+
+ err := b.enableEngineFacadeWithPortCheck(
+ context.Background(),
+ proxy,
+ profile,
+ ollamaHostAlias{},
+ func(int) bool { return true },
+ )
+ if err != nil {
+ t.Fatalf("enable engine facade: %v", err)
+ }
+ first, second := <-attempts, <-attempts
+ if first.Port != profile.FacadePort {
+ t.Fatalf("first port = %d, want stock facade %d", first.Port, profile.FacadePort)
+ }
+ // 1234 is this backend and LM Studio's facade; 1235 is LM Studio's backend.
+ if second.Port != 1236 {
+ t.Fatalf("fallback port = %d, want 1236 after backend and sibling exclusions", second.Port)
+ }
+ if !second.IgnorePersistedPort {
+ t.Fatal("fallback retry could restore the port that just failed")
+ }
+
+ restart := b.defaultEngineFacadeSpec(profile)
+ if restart.Port != second.Port || !restart.IgnorePersistedPort {
+ t.Fatalf("restart spec = %+v, want explicit fallback port %d", restart, second.Port)
+ }
+}
+
+func TestLlamaCPPFacadePreparationPreservesConfiguredPorts(t *testing.T) {
+ profile := mustEngineProxyProfile("llamacpp")
+ for _, tc := range []struct {
+ name string
+ serverPort int
+ proxyPort int
+ explicit bool
+ }{
+ {name: "manifest default", serverPort: profile.EnginePortBase},
+ {name: "explicit settings", serverPort: 18081, proxyPort: 18080, explicit: true},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ b := &Broker{
+ proxyPath: "test-proxy",
+ proxyEngines: []string{profile.Name},
+ engineSettingsLoaded: true,
+ }
+ if tc.explicit {
+ b.engineSettings = map[string]*engineSettingsRecord{
+ profile.Name: {Explicit: true, Snapshot: settings.Snapshot{
+ Settings: settings.Config{ServerPort: tc.serverPort, ProxyPort: tc.proxyPort},
+ }},
+ }
+ }
+ worker, codec := newTestRPCWorkerPipe(t)
+ b.setEngineMgr(worker)
+ var calls atomic.Int32
+ go func() {
+ for {
+ msg, err := codec.Read()
+ if err != nil {
+ return
+ }
+ calls.Add(1)
+ if err := codec.Respond(msg.ID, ollamaPortStatus{Running: true, Port: tc.serverPort}); err != nil {
+ t.Errorf("respond to unexpected engine request %s: %v", msg.Method, err)
+ return
+ }
+ }
+ }()
+
+ b.prepareEnabledFacades()
+
+ if got := calls.Load(); got != 0 {
+ t.Fatalf("preparation issued %d engine requests, want no probing or relocation", got)
+ }
+ state := b.engineProxy(profile)
+ if got := int(state.backendPort.Load()); got != tc.serverPort {
+ t.Fatalf("engine port = %d, want %d", got, tc.serverPort)
+ }
+ if got := int(state.startupPort.Load()); got != tc.proxyPort {
+ t.Fatalf("startup proxy port = %d, want %d", got, tc.proxyPort)
+ }
+ if got := state.explicitSettings.Load(); got != tc.explicit {
+ t.Fatalf("explicit settings = %v, want %v", got, tc.explicit)
+ }
+ })
+ }
+}
+
+func TestLlamaCPPProxyTerminalHandlingPreservesEngineState(t *testing.T) {
+ profile := mustEngineProxyProfile("llamacpp")
+ b := &Broker{ollamaPortReady: make(chan struct{}), lmstudioPortReady: make(chan struct{})}
+ state := b.engineProxy(profile)
+ state.backendPort.Store(int32(profile.EnginePortBase))
+ state.startupPort.Store(18080)
+ state.managedFacade.Store(true)
+
+ b.blockAndFinishEngineProxy(profile)
+ b.finishEngineProxyStartup(profile)
+
+ if got := int(state.backendPort.Load()); got != profile.EnginePortBase {
+ t.Fatalf("engine port = %d, want %d", got, profile.EnginePortBase)
+ }
+ if got := state.startupPort.Load(); got != 18080 || !state.managedFacade.Load() {
+ t.Fatalf("terminal handling changed facade state: port=%d managed=%v", got, state.managedFacade.Load())
+ }
+ for _, gate := range []struct {
+ name string
+ ready <-chan struct{}
+ }{{"Ollama", b.ollamaPortReady}, {"LM Studio", b.lmstudioPortReady}} {
+ select {
+ case <-gate.ready:
+ t.Fatalf("llama.cpp terminal handling released %s's gate", gate.name)
+ default:
+ }
+ }
+}
+
+func TestLlamaCPPEngineStatusRelaysBeforeOtherPortGates(t *testing.T) {
+ profile := mustEngineProxyProfile("llamacpp")
+ b := brokerWithEngineStatus(t, profile.Name, profile.EnginePortBase)
+ b.ollamaPortReady = make(chan struct{})
+ b.lmstudioPortReady = make(chan struct{})
+ b.managedOllamaBackend.Store(managedOllamaBackendStart)
+ client, server := net.Pipe()
+ t.Cleanup(func() {
+ close(b.ollamaPortReady)
+ close(b.lmstudioPortReady)
+ _ = client.Close()
+ _ = server.Close()
+ })
+ b.codec = NewCodec(server)
+ id := json.RawMessage(`1`)
+
+ b.relayToEngine(&Message{ID: &id, Method: "engine:status", Params: json.RawMessage(`{"engine":"llamacpp"}`)})
+
+ if err := client.SetReadDeadline(time.Now().Add(2 * time.Second)); err != nil {
+ t.Fatalf("set response deadline: %v", err)
+ }
+ response, err := NewCodec(client).Read()
+ if err != nil {
+ t.Fatalf("read llama.cpp status while other port gates are pending: %v", err)
+ }
+ if response.Error != nil {
+ t.Fatalf("llama.cpp status failed: %+v", response.Error)
+ }
+ var status ollamaPortStatus
+ if err := json.Unmarshal(response.Result, &status); err != nil {
+ t.Fatalf("decode llama.cpp status: %v", err)
+ }
+ if status.Port != profile.EnginePortBase {
+ t.Fatalf("status port = %d, want %d", status.Port, profile.EnginePortBase)
+ }
+}
+
+func TestLlamaCPPProxyNotificationDispatchPreservesFacadeAddress(t *testing.T) {
+ profile := mustEngineProxyProfile("llamacpp")
+ client, server := net.Pipe()
+ t.Cleanup(func() {
+ _ = client.Close()
+ _ = server.Close()
+ })
+ b := &Broker{codec: NewCodec(server)}
+ b.proxyMu.Lock()
+ b.setEngineProxySubscribed(profile, true)
+ b.proxyMu.Unlock()
+
+ payload := json.RawMessage(`{"port":8080}`)
+ done := make(chan struct{})
+ go func() {
+ b.forwardProxyProcessNotification(0, 0, profile.addressed("ready"), payload)
+ close(done)
+ }()
+ if err := client.SetReadDeadline(time.Now().Add(2 * time.Second)); err != nil {
+ t.Fatalf("set read deadline: %v", err)
+ }
+ msg, err := NewCodec(client).Read()
+ if err != nil {
+ t.Fatalf("read forwarded notification: %v", err)
+ }
+ if msg.Method != profile.ComponentName()+":ready" {
+ t.Fatalf("notification method = %s, want %s:ready", msg.Method, profile.ComponentName())
+ }
+ var ready proxyReadyParams
+ if err := json.Unmarshal(msg.Params, &ready); err != nil {
+ t.Fatalf("decode ready notification: %v", err)
+ }
+ if ready.Port != 8080 {
+ t.Fatalf("ready port = %d, want 8080", ready.Port)
+ }
+ select {
+ case <-done:
+ case <-time.After(2 * time.Second):
+ t.Fatal("notification forwarding did not finish")
+ }
+}
diff --git a/services/nvpair-ui-broker/engineproxy.go b/services/nvpair-ui-broker/engineproxy.go
index 1254d08e..43a5ff91 100644
--- a/services/nvpair-ui-broker/engineproxy.go
+++ b/services/nvpair-ui-broker/engineproxy.go
@@ -9,11 +9,9 @@ package main
// the broker about it. What this file adds is the one thing only the broker
// needs: whether it may reposition the engine's process while it is running.
//
-// That single question decides every place the two engines' port choreography
-// diverges, which is why it is a named enum rather than a set of booleans or a
-// bag of function pointers. A hook would only move the divergent bodies into
-// this file; naming the reason keeps them where they belong and makes the
-// difference reviewable.
+// Ownership governs relocation authority, not facade setup. Engines normally
+// keep their configured port while the broker places their facade. Ollama and
+// LM Studio additionally need engine-specific compatibility-port takeover.
import (
"context"
@@ -30,8 +28,7 @@ import (
"nvpair-shared/noderec"
)
-// engineOwnership answers a single question: may the broker reposition this
-// engine's process while it is running?
+// engineOwnership describes the broker's authority to request engine relocation.
type engineOwnership int
const (
@@ -41,9 +38,10 @@ const (
// it holds the facade port. Ollama.
adoptedEngine engineOwnership = iota
- // managedEngine — engine-manager launched it in identified command mode
- // and has an official stop command for it, so it may be stopped and
- // repositioned before the proxy starts. LM Studio.
+ // managedEngine — engine-manager can stop and reposition a process it
+ // owns, or an identified command-mode runtime with an official stop
+ // command. It still refuses unknown or unowned processes. LM Studio and
+ // llama.cpp; only LM Studio needs automatic compatibility-port takeover.
managedEngine
)
@@ -52,9 +50,8 @@ const (
type engineProxyProfile struct {
engines.Engine
- // Ownership decides the port choreography: plan-then-commit for an
- // adopted engine, move-then-verify for a managed one. It is the only
- // judgment call in adding an engine.
+ // Ownership determines whether a running engine may be relocated when an
+ // engine-specific compatibility-port takeover requires it.
Ownership engineOwnership
// HealthProbePath is the path whose 200 means "this engine is answering".
@@ -150,9 +147,9 @@ func (b *Broker) engineProxy(p engineProxyProfile) *engineProxyRuntime {
return b.engineProxies[p.Name]
}
-// ollamaState and lmstudioState are shorthand for the two engines this build
-// ships, for code that is inherently about one of them. Profile-generic code
-// should take an engineProxyProfile and call engineProxy instead.
+// ollamaState and lmstudioState are shorthands for code that is inherently
+// about those engines' special ownership behavior. Profile-generic code should
+// take an engineProxyProfile and call engineProxy instead.
func (b *Broker) ollamaState() *engineProxyRuntime { return b.engineProxy(ollamaProxyProfile) }
func (b *Broker) lmstudioState() *engineProxyRuntime { return b.engineProxy(lmstudioProxyProfile) }
@@ -164,9 +161,10 @@ var engineProxyProfiles = buildEngineProxyProfiles()
func buildEngineProxyProfiles() []engineProxyProfile {
brokerOnly := map[string]engineProxyProfile{
"ollama": {Ownership: adoptedEngine, HealthProbePath: "/"},
- // LM Studio is the one engine engine-manager may move while running:
- // its identified command-mode runtime has an official stop command.
+ // LM Studio's command-mode runtime has an official stop command;
+ // llama.cpp's managed process is stopped directly by engine-manager.
"lmstudio": {Ownership: managedEngine, HealthProbePath: "/v1/models"},
+ "llamacpp": {Ownership: managedEngine, HealthProbePath: "/health"},
}
out := make([]engineProxyProfile, 0, len(engines.All()))
for _, e := range engines.All() {
@@ -295,6 +293,34 @@ func (b *Broker) enableProxyFacadeWithFallback(
return b.enableProxyFacade(parent, p, spec)
}
+// defaultEngineFacadeSpec prefers the stock facade port, while preserving a
+// fallback or explicit port already selected for this broker lifetime.
+func (b *Broker) defaultEngineFacadeSpec(profile engineProxyProfile) enableFacadeRequest {
+ spec := enableFacadeRequest{Engine: profile.Name, Port: profile.FacadePort}
+ if port := int(b.engineProxy(profile).startupPort.Load()); port != 0 {
+ spec.Port = port
+ spec.IgnorePersistedPort = true
+ }
+ return spec
+}
+
+// defaultEngineFallbackPortWithCheck keeps a fallback off the engine's default
+// port and every sibling's facade/backend/persisted ports.
+func (b *Broker) defaultEngineFallbackPortWithCheck(
+ profile engineProxyProfile, failed int, available func(int) bool,
+) int {
+ excluded := []int{failed, profile.EnginePortBase}
+ if alias := b.currentOllamaHostAlias().Port; alias > 0 {
+ excluded = append(excluded, alias)
+ }
+ for port := range b.siblingEngineProxyPorts(profile) {
+ excluded = append(excluded, port)
+ }
+ fallback := nextAvailablePortExcluding(profile.FacadePort, excluded, available)
+ b.engineProxy(profile).startupPort.Store(int32(fallback))
+ return fallback
+}
+
// facadeMethodFor strips a facade-scoped notification's engine address and
// confirms it belongs to the engine this reader speaks for.
//
@@ -361,24 +387,38 @@ func (b *Broker) proxyEnabled(p engineProxyProfile) bool {
return false
}
-// prepareEnabledFacades prepares managed port ownership for the engines the
-// broker is actually going to front, in the table's order — Ollama first,
-// because its preparation reserves any inherited OLLAMA_HOST alias that later
-// engines must route around.
+// prepareEnabledFacades prepares port ownership for the engines the broker is
+// actually going to front, in table order — Ollama first, because its alias
+// reservation constrains later engines. The default path records the engine
+// port without a gate or automatic relocation.
//
// The enablement check belongs here and not downstream, because preparation is
-// not read-only: for a managed engine the backend move runs inside it, so
+// not read-only: LM Studio's engine move runs inside it, so
// preparing an engine whose proxy is never started relocates that engine off
// its own stock port and leaves nothing serving it. Ollama cannot show the
// symptom, since its move is deferred until its proxy proves it holds the
// facade — which is exactly why this cannot be left to the callee.
func (b *Broker) prepareEnabledFacades() {
- if b.proxyEnabled(ollamaProxyProfile) {
- b.prepareManagedOllamaFacade()
+ for _, profile := range engineProxyProfiles {
+ if !b.proxyEnabled(profile) {
+ continue
+ }
+ switch profile.Name {
+ case ollamaProxyProfile.Name:
+ b.prepareManagedOllamaFacade()
+ case lmstudioProxyProfile.Name:
+ b.prepareManagedLMStudioFacade()
+ default:
+ b.prepareDefaultEngineFacade(profile)
+ }
}
- if b.proxyEnabled(lmstudioProxyProfile) {
- b.prepareManagedLMStudioFacade()
+}
+
+func (b *Broker) prepareDefaultEngineFacade(profile engineProxyProfile) {
+ if b.prepareExplicitEngineSettings(profile.Name) {
+ return
}
+ b.engineProxy(profile).backendPort.Store(int32(profile.EnginePortBase))
}
// proxyDisabledReason explains why an engine has no proxy, and reports whether
@@ -452,10 +492,6 @@ func (b *Broker) setEngineProxyHandle(p engineProxyProfile, proxy *proxyProcess)
//
// This is the end of the line for a notification — every path consumes it, so
// there is nothing for a caller to do afterwards and nothing to report back.
-// The heads of the two callers stay separate: the bind-failure and readiness
-// handling genuinely differ by ownership, and folding them in behind a
-// callback would move those bodies into this file without making them any more
-// shared.
func (b *Broker) forwardEngineProxyNotification(profile engineProxyProfile, method string, params json.RawMessage) {
if b.routeProcessScopedProxyNotification(method, params) {
return
@@ -472,6 +508,19 @@ func (b *Broker) forwardEngineProxyNotification(profile engineProxyProfile, meth
}
}
+// forwardDefaultEngineProxyNotification relays a facade notification without
+// engine-specific compatibility-port reconciliation.
+func (b *Broker) forwardDefaultEngineProxyNotification(profile engineProxyProfile, method string, params json.RawMessage) {
+ method, addressed := facadeMethodFor(profile, method)
+ if !addressed {
+ return
+ }
+ if b.dispatchErrorsNotif(profile.ComponentName(), method, params) {
+ return
+ }
+ b.forwardEngineProxyNotification(profile, method, params)
+}
+
// routeProcessScopedProxyNotification handles the notifications that belong to
// the proxy process rather than to one of its facades, and reports whether it
// took the method.
@@ -509,6 +558,58 @@ func (b *Broker) setEngineProxySubscribed(p engineProxyProfile, subscribed bool)
return was
}
+// handleEngineProxyBrokerRequest serves the facade methods owned by the broker
+// rather than the proxy child. It reports whether method was handled.
+func (b *Broker) handleEngineProxyBrokerRequest(profile engineProxyProfile, method string, msg *Message) bool {
+ switch method {
+ case "set-port":
+ // Settings application round-trips through worker readers, so it
+ // must not block the broker's JSON-RPC read pump.
+ go b.handleSettingsPortRPC(msg, profile.Name)
+ return true
+
+ case "get-status":
+ var result ProxyStatusResult
+ if proxy := b.engineProxyHandle(profile); proxy != nil {
+ result.Ready, result.Port = proxy.Status(profile.Name)
+ }
+ if err := b.codec.Respond(msg.ID, result); err != nil {
+ log.Printf("failed to respond to %s:get-status: %v", profile.ComponentName(), err)
+ }
+ return true
+
+ case "subscribe":
+ b.proxyMu.Lock()
+ wasSubscribed := b.setEngineProxySubscribed(profile, true)
+ b.proxyMu.Unlock()
+ if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: true}); err != nil {
+ log.Printf("failed to respond to %s:subscribe: %v", profile.ComponentName(), err)
+ }
+ // The acknowledgement must precede the baseline notification. A
+ // redundant subscription is already live and needs no replay.
+ if !wasSubscribed {
+ if proxy := b.engineProxyHandle(profile); proxy != nil {
+ if params := proxy.ReadyParams(profile.Name); params != nil {
+ if err := b.codec.Notify(profile.ComponentName()+":ready", params); err != nil {
+ slog.Warn("emit baseline proxy ready failed", "engine", profile.Name, "err", err)
+ }
+ }
+ }
+ }
+ return true
+
+ case "unsubscribe":
+ b.proxyMu.Lock()
+ b.setEngineProxySubscribed(profile, false)
+ b.proxyMu.Unlock()
+ if err := b.codec.Respond(msg.ID, SubscriptionResult{Subscribed: false}); err != nil {
+ log.Printf("failed to respond to %s:unsubscribe: %v", profile.ComponentName(), err)
+ }
+ return true
+ }
+ return false
+}
+
// relayToEngineProxy forwards an -proxy: client request to that
// engine's facade and maps the response straight back.
//
diff --git a/services/nvpair-ui-broker/engineproxy_test.go b/services/nvpair-ui-broker/engineproxy_test.go
index 7319e55e..cac620c8 100644
--- a/services/nvpair-ui-broker/engineproxy_test.go
+++ b/services/nvpair-ui-broker/engineproxy_test.go
@@ -46,6 +46,7 @@ func TestEngineHealthProbePaths(t *testing.T) {
}{
{"ollama", "/"},
{"lmstudio", "/v1/models"},
+ {"llamacpp", "/health"},
} {
p, ok := engineProxyProfileFor(tc.engine)
if !ok {
@@ -92,8 +93,8 @@ func TestBrokerConstantsMatchTheEngineTable(t *testing.T) {
}
}
-// Ownership is the one judgment call in adding an engine, so the two values in
-// the table today are pinned explicitly. Getting these backwards does not fail
+// Every engine's relocation authority is pinned explicitly. Getting these
+// backwards does not fail
// to compile — it silently changes which engine the broker believes it may stop.
func TestEngineOwnershipAssignments(t *testing.T) {
for _, tc := range []struct {
@@ -102,6 +103,7 @@ func TestEngineOwnershipAssignments(t *testing.T) {
}{
{"ollama", adoptedEngine},
{"lmstudio", managedEngine},
+ {"llamacpp", managedEngine},
} {
p, ok := engineProxyProfileFor(tc.engine)
if !ok {
@@ -167,7 +169,7 @@ func TestParseProxyEngines(t *testing.T) {
want []string
wantErr bool
}{
- {name: "default is every engine", csv: "ollama,lmstudio", want: []string{"ollama", "lmstudio"}},
+ {name: "default set", csv: "ollama,lmstudio,llamacpp", want: []string{"ollama", "lmstudio", "llamacpp"}},
{name: "single engine", csv: "lmstudio", want: []string{"lmstudio"}},
{name: "whitespace and blanks are tolerated", csv: " ollama , , lmstudio ", want: []string{"ollama", "lmstudio"}},
{name: "duplicates collapse", csv: "ollama,ollama", want: []string{"ollama"}},
diff --git a/services/nvpair-ui-broker/enginesettings.go b/services/nvpair-ui-broker/enginesettings.go
index 9f65e88b..de6d2bbe 100644
--- a/services/nvpair-ui-broker/enginesettings.go
+++ b/services/nvpair-ui-broker/enginesettings.go
@@ -18,6 +18,7 @@ import (
"nvpair-shared/appdir"
"nvpair-shared/clustertrust"
+ "nvpair-shared/engines"
settings "nvpair-shared/enginesettings"
"nvpair-shared/noderec"
)
@@ -160,17 +161,24 @@ func (b *Broker) settingsWorkerCall(ctx context.Context, method string, params a
}
func (b *Broker) settingsProxy(engine string) *proxyProcess {
- if engine == "ollama" {
- return b.getProxy()
+ profile, ok := engineProxyProfileFor(engine)
+ if !ok {
+ return nil
}
- if engine == "lmstudio" {
- return b.getLMStudioProxy()
+ return b.engineProxyHandle(profile)
+}
+
+func isEngineDiscoveryService(service noderec.ServiceKey) bool {
+ for _, profile := range engineProxyProfiles {
+ if profile.DiscoveryService == service {
+ return true
+ }
}
- return nil
+ return false
}
func (b *Broker) settingsSnapshotLocked(ctx context.Context, engine string) (settings.Snapshot, error) {
- if engine != "ollama" && engine != "lmstudio" {
+ if _, ok := engineProxyProfileFor(engine); !ok {
return settings.Snapshot{}, fmt.Errorf("this engine does not support settings")
}
if err := b.loadEngineSettingsLocked(); err != nil {
@@ -233,8 +241,8 @@ func (b *Broker) settingsSnapshotLocked(ctx context.Context, engine string) (set
func (b *Broker) publishSettingsLocked() {
all := make([]settings.Snapshot, 0, len(b.engineSettings))
- for _, engine := range []string{"ollama", "lmstudio"} {
- if record := b.engineSettings[engine]; record != nil {
+ for _, profile := range engineProxyProfiles {
+ if record := b.engineSettings[profile.Name]; record != nil {
record.Snapshot.Sequence++
all = append(all, record.Snapshot)
if b.codec != nil {
@@ -261,7 +269,7 @@ func (b *Broker) validateSettingsPortsLocked(ctx context.Context, engine string,
// Include every registered PAIR listener, including services added later.
if b.regCache != nil {
for _, service := range b.regCache.Snapshot() {
- if service.Port > 0 && service.Service != noderec.ServiceOllama && service.Service != noderec.ServiceLMStudio {
+ if service.Port > 0 && !isEngineDiscoveryService(service.Service) {
reserved[service.Port] = true
}
}
@@ -283,18 +291,18 @@ func (b *Broker) validateSettingsPortsLocked(ctx context.Context, engine string,
return fmt.Errorf("a selected port is reserved by another configured engine")
}
}
- for _, other := range []string{"ollama", "lmstudio"} {
- if other == engine {
+ for _, profile := range engineProxyProfiles {
+ if profile.Name == engine {
continue
}
- if record := b.engineSettings[other]; record != nil {
+ if record := b.engineSettings[profile.Name]; record != nil {
port := record.Snapshot.Settings.ProxyPort
if port == config.ServerPort || port == config.ProxyPort {
return fmt.Errorf("a selected port is reserved by another configured proxy")
}
}
- if proxy := b.settingsProxy(other); proxy != nil {
- _, port := proxy.Status(other)
+ if proxy := b.settingsProxy(profile.Name); proxy != nil {
+ _, port := proxy.Status(profile.Name)
if port == config.ServerPort || port == config.ProxyPort {
return fmt.Errorf("a selected port is already used by another proxy")
}
@@ -350,10 +358,8 @@ func (b *Broker) runSettingsOperationLocked(ctx context.Context, engine string,
if err := b.migrateSettingsArgumentsLocked(ctx, engine, record); err != nil {
return err
}
- service := noderec.ServiceOllama
- if engine == "lmstudio" {
- service = noderec.ServiceLMStudio
- }
+ profile, _ := engineProxyProfileFor(engine)
+ service := profile.DiscoveryService
if b.regCache != nil {
b.unregisterService(service)
}
@@ -398,11 +404,7 @@ func (b *Broker) runSettingsOperationLocked(ctx context.Context, engine string,
record.Snapshot.EffectiveServerPort = launch.EffectivePort
record.Snapshot.Editable = launch.Editable
record.Snapshot.Reason = launch.Reason
- if engine == "ollama" {
- b.ollamaState().backendPort.Store(int32(launch.EffectivePort))
- } else {
- b.lmstudioState().backendPort.Store(int32(launch.EffectivePort))
- }
+ b.engineProxy(profile).backendPort.Store(int32(launch.EffectivePort))
}
proxyReady := false
if proxy := b.settingsProxy(engine); proxy != nil {
@@ -683,12 +685,9 @@ func (b *Broker) rebindSettingsProxy(engine string, port int) error {
// Disable automatic facade takeover before the ready event can race the
// explicit rebind. The accepted journal restores these choices after restart.
profile, _ := engineProxyProfileFor(engine)
- b.engineProxy(profile).explicitSettings.Store(true)
- if engine == "ollama" {
- b.ollamaState().managedFacade.Store(false)
- } else {
- b.lmstudioState().managedFacade.Store(false)
- }
+ state := b.engineProxy(profile)
+ state.explicitSettings.Store(true)
+ state.managedFacade.Store(false)
_, rpcErr, err := p.Call(context.Background(), engine+":set-port", settingsJSON(map[string]int{"port": port}))
if err != nil {
return err
@@ -696,13 +695,8 @@ func (b *Broker) rebindSettingsProxy(engine string, port int) error {
if rpcErr != nil {
return fmt.Errorf("%s", rpcErr.Message)
}
- if engine == "ollama" {
- b.ollamaState().managedFacade.Store(false)
- b.ollamaState().startupPort.Store(int32(port))
- } else {
- b.lmstudioState().managedFacade.Store(false)
- b.lmstudioState().startupPort.Store(int32(port))
- }
+ state.managedFacade.Store(false)
+ state.startupPort.Store(int32(port))
return nil
}
@@ -723,7 +717,7 @@ func (b *Broker) refreshEngineSettings(ctx context.Context) {
readCtx, cancel := context.WithTimeout(ctx, 5*time.Second)
defer cancel()
changed := false
- for _, engine := range []string{"ollama", "lmstudio"} {
+ for _, engine := range engines.Names() {
var before settings.Snapshot
if r := b.engineSettings[engine]; r != nil {
before = r.Snapshot
diff --git a/services/nvpair-ui-broker/enginesettings_recovery.go b/services/nvpair-ui-broker/enginesettings_recovery.go
index ec444c39..dcaba5b1 100644
--- a/services/nvpair-ui-broker/enginesettings_recovery.go
+++ b/services/nvpair-ui-broker/enginesettings_recovery.go
@@ -73,17 +73,14 @@ func (b *Broker) prepareExplicitEngineSettings(engine string) bool {
return false
}
profile, _ := engineProxyProfileFor(engine)
- b.engineProxy(profile).explicitSettings.Store(true)
+ state := b.engineProxy(profile)
+ state.explicitSettings.Store(true)
+ state.managedFacade.Store(false)
+ state.backendPort.Store(int32(config.ServerPort))
+ state.startupPort.Store(int32(config.ProxyPort))
if engine == "ollama" {
- b.ollamaState().managedFacade.Store(false)
b.managedOllamaBackend.Store(0)
- b.ollamaState().backendPort.Store(int32(config.ServerPort))
- b.ollamaState().startupPort.Store(int32(config.ProxyPort))
b.syncCurrentEngineOllamaHostAliasReservation()
- } else {
- b.lmstudioState().managedFacade.Store(false)
- b.lmstudioState().backendPort.Store(int32(config.ServerPort))
- b.lmstudioState().startupPort.Store(int32(config.ProxyPort))
}
return true
}
diff --git a/services/nvpair-ui-broker/enginesettings_review_test.go b/services/nvpair-ui-broker/enginesettings_review_test.go
index 50cb11be..41b66d8b 100644
--- a/services/nvpair-ui-broker/enginesettings_review_test.go
+++ b/services/nvpair-ui-broker/enginesettings_review_test.go
@@ -21,8 +21,10 @@ func TestSettingsRebindAddressesOnlyRequestedFacade(t *testing.T) {
t.Run(profile.Name, func(t *testing.T) {
h := newSettingsHarness(t)
p := h.b.getProxy()
- _, ollamaBefore := p.Status("ollama")
- _, lmstudioBefore := p.Status("lmstudio")
+ before := make(map[string]int, len(engineProxyProfiles))
+ for _, candidate := range engineProxyProfiles {
+ _, before[candidate.Name] = p.Status(candidate.Name)
+ }
ln, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatal(err)
@@ -32,13 +34,8 @@ func TestSettingsRebindAddressesOnlyRequestedFacade(t *testing.T) {
if err := h.b.rebindSettingsProxy(profile.Name, port); err != nil {
t.Fatal(err)
}
- wantOllama, wantLMStudio := ollamaBefore, lmstudioBefore
- if profile.Name == "ollama" {
- wantOllama = port
- } else {
- wantLMStudio = port
- }
- for engine, want := range map[string]int{"ollama": wantOllama, "lmstudio": wantLMStudio} {
+ before[profile.Name] = port
+ for engine, want := range before {
ready, got := p.Status(engine)
if !ready || got != want {
t.Fatalf("%s ready=%v port=%d, want %d", engine, ready, got, want)
@@ -61,10 +58,13 @@ func TestExplicitSettingsBindFailurePreservesChosenPort(t *testing.T) {
t.Fatal("explicit settings were not restored")
}
failure := settingsJSON(map[string]any{"code": "bind-failed", "port": requested})
- if profile.Name == "ollama" {
+ switch profile.Name {
+ case "ollama":
b.forwardProxyNotification("error", failure)
- } else {
+ case "lmstudio":
b.forwardLMStudioProxyNotification("error", failure)
+ default:
+ b.forwardDefaultEngineProxyNotification(profile, profile.addressed("error"), failure)
}
if got := b.engineProxy(profile).startupPort.Load(); got != requested {
t.Fatalf("bind notification changed chosen port to %d", got)
diff --git a/services/nvpair-ui-broker/enginesettings_test.go b/services/nvpair-ui-broker/enginesettings_test.go
index e8d487b6..6dc1e892 100644
--- a/services/nvpair-ui-broker/enginesettings_test.go
+++ b/services/nvpair-ui-broker/enginesettings_test.go
@@ -16,6 +16,7 @@ import (
"testing"
"time"
+ "nvpair-shared/engines"
settings "nvpair-shared/enginesettings"
"nvpair-shared/noderec"
"nvpair-ui-broker/relay"
@@ -24,6 +25,7 @@ import (
type settingsHarness struct {
b *Broker
applies atomic.Int32
+ proxyRebinds atomic.Int32
fail atomic.Bool
failBeforeStop atomic.Bool
loseProxyOnStop atomic.Bool
@@ -38,7 +40,12 @@ type settingsHarness struct {
func newSettingsHarness(t *testing.T) *settingsHarness {
t.Helper()
- ports := make([]int, 3)
+ return newSettingsHarnessForEngine(t, "ollama")
+}
+
+func newSettingsHarnessForEngine(t *testing.T, engine string) *settingsHarness {
+ t.Helper()
+ ports := make([]int, 4)
listeners := []net.Listener{}
for i := range ports {
ln, err := net.Listen("tcp", "127.0.0.1:0")
@@ -58,9 +65,11 @@ func newSettingsHarness(t *testing.T) *settingsHarness {
proxy := &proxyProcess{peer: proxyWorker.peer, facadeState: map[string]proxyFacadeState{
"ollama": {ready: true, port: ports[1]},
"lmstudio": {ready: true, port: ports[2]},
+ "llamacpp": {ready: true, port: ports[3]},
}}
- h.b.setProxy(proxy)
- h.b.setLMStudioProxy(proxy)
+ for _, profile := range engineProxyProfiles {
+ h.b.setEngineProxyHandle(profile, proxy)
+ }
go func() {
for {
msg, err := proxyCodec.Read()
@@ -70,25 +79,26 @@ func newSettingsHarness(t *testing.T) *settingsHarness {
if !msg.IsRequest() {
continue
}
- if msg.Method != "ollama:set-port" && msg.Method != "lmstudio:set-port" {
+ engine, method := engines.SplitAddressedMethod(msg.Method)
+ if engine == "" || method != "set-port" {
_ = proxyCodec.Respond(msg.ID, map[string]bool{"ok": true})
continue
}
var p struct {
Port int `json:"port"`
}
- _ = json.Unmarshal(msg.Params, &p)
- proxy.readyMu.Lock()
- engine := "ollama"
- if msg.Method == "lmstudio:set-port" {
- engine = "lmstudio"
+ if err := json.Unmarshal(msg.Params, &p); err != nil {
+ t.Errorf("decode proxy port request: %v", err)
+ return
}
+ h.proxyRebinds.Add(1)
+ proxy.readyMu.Lock()
proxy.facadeState[engine] = proxyFacadeState{ready: true, port: p.Port}
proxy.readyMu.Unlock()
_ = proxyCodec.Respond(msg.ID, map[string]int{"port": p.Port})
}
}()
- launch := settings.LaunchState{Engine: "ollama", ServerPort: ports[0], EffectivePort: ports[0], LaunchText: "--fixture-option", Running: true, Editable: true, Format: "pair-arguments-v1"}
+ launch := settings.LaunchState{Engine: engine, ServerPort: ports[0], EffectivePort: ports[0], LaunchText: "--fixture-option", Running: true, Editable: true, Format: "pair-arguments-v1"}
go func() {
for {
msg, err := codec.Read()
@@ -130,9 +140,9 @@ func newSettingsHarness(t *testing.T) *settingsHarness {
if h.failBeforeStop.Load() {
if h.loseProxyOnStop.Load() {
proxy.readyMu.Lock()
- state := proxy.facadeState["ollama"]
+ state := proxy.facadeState[engine]
state.ready = false
- proxy.facadeState["ollama"] = state
+ proxy.facadeState[engine] = state
proxy.readyMu.Unlock()
}
_ = codec.RespondError(msg.ID, -32000, "stop failure")
diff --git a/services/nvpair-ui-broker/health_connections_test.go b/services/nvpair-ui-broker/health_connections_test.go
index 74e7231a..da685cba 100644
--- a/services/nvpair-ui-broker/health_connections_test.go
+++ b/services/nvpair-ui-broker/health_connections_test.go
@@ -51,4 +51,5 @@ func TestHealthChecksReuseConnections(t *testing.T) {
test("ollama", "/", ollamaProxyProfile)
test("lmstudio", "/v1/models", lmstudioProxyProfile)
+ test("llamacpp", "/health", mustEngineProxyProfile("llamacpp"))
}
diff --git a/services/nvpair-ui-broker/proxyaddressing_test.go b/services/nvpair-ui-broker/proxyaddressing_test.go
index 533b1769..9b5192b8 100644
--- a/services/nvpair-ui-broker/proxyaddressing_test.go
+++ b/services/nvpair-ui-broker/proxyaddressing_test.go
@@ -234,6 +234,89 @@ func TestReadinessIsTrackedPerEngine(t *testing.T) {
}
}
+func TestBrokerOwnedFacadeMethodsFollowTheProfile(t *testing.T) {
+ for _, profile := range engineProxyProfiles {
+ t.Run(profile.Name, func(t *testing.T) {
+ client, server := net.Pipe()
+ t.Cleanup(func() {
+ _ = client.Close()
+ _ = server.Close()
+ })
+ payload := json.RawMessage(fmt.Sprintf(`{"version":"test","port":%d}`, profile.FacadePort))
+ proxy := &proxyProcess{facadeState: map[string]proxyFacadeState{
+ profile.Name: {
+ ready: true,
+ port: profile.FacadePort,
+ params: payload,
+ },
+ }}
+ b := &Broker{codec: NewCodec(server)}
+ b.setEngineProxyHandle(profile, proxy)
+ reader := NewCodec(client)
+ nextID := 0
+ call := func(method string, frameCount int) []*Message {
+ t.Helper()
+ nextID++
+ id := json.RawMessage(fmt.Sprintf("%d", nextID))
+ done := make(chan struct{})
+ go func() {
+ b.handleMessage(&Message{JSONRPC: "2.0", ID: &id, Method: profile.ComponentName() + ":" + method})
+ close(done)
+ }()
+ frames := make([]*Message, 0, frameCount)
+ for range frameCount {
+ if err := client.SetReadDeadline(time.Now().Add(2 * time.Second)); err != nil {
+ t.Fatalf("set read deadline: %v", err)
+ }
+ frame, err := reader.Read()
+ if err != nil {
+ t.Fatalf("read %s frame: %v", method, err)
+ }
+ frames = append(frames, frame)
+ }
+ select {
+ case <-done:
+ case <-time.After(2 * time.Second):
+ t.Fatalf("%s handler did not finish after %d frame(s)", method, frameCount)
+ }
+ return frames
+ }
+
+ statusFrames := call("get-status", 1)
+ var status ProxyStatusResult
+ if err := json.Unmarshal(statusFrames[0].Result, &status); err != nil {
+ t.Fatalf("decode status: %v", err)
+ }
+ if !status.Ready || status.Port != profile.FacadePort {
+ t.Fatalf("status = %+v, want ready on %d", status, profile.FacadePort)
+ }
+
+ subscribeFrames := call("subscribe", 2)
+ var subscribed SubscriptionResult
+ if err := json.Unmarshal(subscribeFrames[0].Result, &subscribed); err != nil || !subscribed.Subscribed {
+ t.Fatalf("subscribe response = %s, error %v", subscribeFrames[0].Result, err)
+ }
+ if subscribeFrames[1].Method != profile.ComponentName()+":ready" {
+ t.Fatalf("second subscribe frame = %q, want ready baseline after response", subscribeFrames[1].Method)
+ }
+ if string(subscribeFrames[1].Params) != string(payload) {
+ t.Fatalf("ready baseline = %s, want %s", subscribeFrames[1].Params, payload)
+ }
+
+ unsubscribeFrames := call("unsubscribe", 1)
+ if err := json.Unmarshal(unsubscribeFrames[0].Result, &subscribed); err != nil || subscribed.Subscribed {
+ t.Fatalf("unsubscribe response = %s, error %v", unsubscribeFrames[0].Result, err)
+ }
+ b.proxyMu.Lock()
+ stillSubscribed := b.engineProxySubscribed(profile)
+ b.proxyMu.Unlock()
+ if stillSubscribed {
+ t.Fatal("facade remained subscribed after unsubscribe")
+ }
+ })
+ }
+}
+
// Each facade subscribes for its own engine's discovery service, so the broker
// has to track a subscription per engine. A single id per process let the second
// facade's subscribe replace the first's, which unsubscribed a live facade and
diff --git a/services/nvpair-ui-broker/proxyport.go b/services/nvpair-ui-broker/proxyport.go
index 7ecfb7b5..298a3c02 100644
--- a/services/nvpair-ui-broker/proxyport.go
+++ b/services/nvpair-ui-broker/proxyport.go
@@ -183,7 +183,7 @@ func (b *Broker) finishEngineProxyStartup(profile engineProxyProfile) {
case lmstudioProxyProfile.Name:
b.finishLMStudioProxyTerminal()
default:
- slog.Warn("no startup-gate finisher for engine", "engine", profile.Name)
+ // Other engines have no compatibility-port startup gate.
}
}
diff --git a/services/nvpair-ui-broker/settingsport.go b/services/nvpair-ui-broker/settingsport.go
index c1d11105..763c5065 100644
--- a/services/nvpair-ui-broker/settingsport.go
+++ b/services/nvpair-ui-broker/settingsport.go
@@ -12,7 +12,7 @@ import (
)
// handleSettingsPortRPC serves the port-only RPCs — engine:set-port,
-// proxy:set-port, and lmstudio-proxy:set-port — which nvpair-tui calls to move
+// -proxy:set-port — which nvpair-tui calls to move
// a single port without rendering the full launch settings form. They keep
// their own narrow request and response shapes, but run through the same
// authoritative settings operation as the desktop editor, so a port change
diff --git a/services/nvpair-ui-broker/settingsport_test.go b/services/nvpair-ui-broker/settingsport_test.go
new file mode 100644
index 00000000..538a3898
--- /dev/null
+++ b/services/nvpair-ui-broker/settingsport_test.go
@@ -0,0 +1,186 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package main
+
+import (
+ "context"
+ "encoding/json"
+ "net"
+ "os"
+ "strings"
+ "testing"
+ "time"
+
+ settings "nvpair-shared/enginesettings"
+)
+
+func callBrokerPortRequest(t *testing.T, b *Broker, method string, params json.RawMessage) *Message {
+ t.Helper()
+ client, server := net.Pipe()
+ t.Cleanup(func() { _ = client.Close(); _ = server.Close() })
+ if err := client.SetReadDeadline(time.Now().Add(5 * time.Second)); err != nil {
+ t.Fatalf("set response deadline: %v", err)
+ }
+ b.codec = NewCodec(server)
+ id := json.RawMessage(`1`)
+ go b.handleMessage(&Message{JSONRPC: "2.0", ID: &id, Method: method, Params: params})
+ codec := NewCodec(client)
+ for {
+ response, err := codec.Read()
+ if err != nil {
+ t.Fatalf("read %s response: %v", method, err)
+ }
+ if response.IsNotification() {
+ continue
+ }
+ if response.ID == nil || string(*response.ID) != string(id) {
+ t.Fatalf("unexpected response ID: %+v", response)
+ }
+ return response
+ }
+}
+
+func TestBrokerProxySetPortRejectsInvalidPorts(t *testing.T) {
+ for _, profile := range engineProxyProfiles {
+ t.Run(profile.Name, func(t *testing.T) {
+ for _, tc := range []struct{ name, params string }{
+ {"malformed parameters", `{`},
+ {"missing port", `{}`},
+ {"zero port", `{"port":0}`},
+ {"negative port", `{"port":-1}`},
+ {"oversized port", `{"port":65536}`},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ response := callBrokerPortRequest(t, &Broker{}, profile.ComponentName()+":set-port", json.RawMessage(tc.params))
+ if response.Error == nil || response.Error.Code != -32602 || response.Error.Message != "port must be between 1 and 65535" {
+ t.Fatalf("response = %+v, want invalid-port error", response)
+ }
+ })
+ }
+ })
+ }
+}
+
+func TestBrokerProxySetPortRejectsInheritedAlias(t *testing.T) {
+ for _, profile := range engineProxyProfiles {
+ t.Run(profile.Name, func(t *testing.T) {
+ b := &Broker{}
+ b.setOllamaHostAlias(ollamaHostAlias{Port: 11433})
+ response := callBrokerPortRequest(t, b, profile.ComponentName()+":set-port", json.RawMessage(`{"port":11433}`))
+ if response.Error == nil || response.Error.Code != -32000 || !strings.Contains(response.Error.Message, "OLLAMA_HOST proxy alias") {
+ t.Fatalf("response = %+v, want alias-port rejection", response)
+ }
+ })
+ }
+}
+
+func TestBrokerLlamaCPPProxySetPortRejectsConflicts(t *testing.T) {
+ test := func(name string, port func(*testing.T, *settingsHarness, settings.Snapshot) int) {
+ t.Run(name, func(t *testing.T) {
+ h := newSettingsHarnessForEngine(t, "llamacpp")
+ before, err := h.b.getEngineSettings(context.Background(), settings.Request{Engine: "llamacpp"}, "")
+ if err != nil {
+ t.Fatalf("read initial settings: %v", err)
+ }
+ target := port(t, h, before)
+ response := callBrokerPortRequest(t, h.b, "llamacpp-proxy:set-port", settingsJSON(map[string]int{"port": target}))
+ if response.Error == nil || response.Error.Code != -32000 || response.Error.Message != "resolve settings errors and port conflicts before applying" {
+ t.Fatalf("error = %+v, want settings-conflict rejection", response.Error)
+ }
+ if h.applies.Load() != 0 || h.proxyRebinds.Load() != 0 {
+ t.Fatal("rejected port request changed runtime")
+ }
+ h.b.engineConfigMu.Lock()
+ after := h.b.engineSettings["llamacpp"].Snapshot
+ h.b.engineConfigMu.Unlock()
+ if after.Settings != before.Settings || after.Revision != before.Revision {
+ t.Fatalf("rejection changed desired settings: before=%+v after=%+v", before, after)
+ }
+ })
+ }
+ test("PAIR service port", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int {
+ return engineControlPort
+ })
+ test("same engine server port", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int {
+ return before.Settings.ServerPort
+ })
+ test("another configured engine", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int {
+ h.otherEnginePort.Store(25001)
+ return 25001
+ })
+ test("another proxy listener", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int {
+ _, port := h.b.getProxy().Status("ollama")
+ return port
+ })
+ test("occupied listener", func(t *testing.T, h *settingsHarness, before settings.Snapshot) int {
+ ln, err := net.Listen("tcp", ":0")
+ if err != nil {
+ t.Fatalf("bind occupied port: %v", err)
+ }
+ t.Cleanup(func() { _ = ln.Close() })
+ return ln.Addr().(*net.TCPAddr).Port
+ })
+}
+
+func TestBrokerLlamaCPPProxySetPortPersistsOnlyRequestedFacade(t *testing.T) {
+ h := newSettingsHarnessForEngine(t, "llamacpp")
+ before, err := h.b.getEngineSettings(context.Background(), settings.Request{Engine: "llamacpp"}, "")
+ if err != nil {
+ t.Fatalf("read initial settings: %v", err)
+ }
+ otherPorts := make(map[string]int)
+ for _, engine := range []string{"ollama", "lmstudio"} {
+ _, otherPorts[engine] = h.b.getProxy().Status(engine)
+ }
+ ln, err := net.Listen("tcp", "127.0.0.1:0")
+ if err != nil {
+ t.Fatalf("allocate target port: %v", err)
+ }
+ target := ln.Addr().(*net.TCPAddr).Port
+ if err := ln.Close(); err != nil {
+ t.Fatalf("release target port: %v", err)
+ }
+ // The namespace determines the engine, even if parameters name another.
+ response := callBrokerPortRequest(t, h.b, "llamacpp-proxy:set-port", settingsJSON(map[string]any{"engine": "ollama", "port": target}))
+ if response.Error != nil {
+ t.Fatalf("set proxy port: %+v", response.Error)
+ }
+ var result struct {
+ Port int `json:"port"`
+ }
+ if err := json.Unmarshal(response.Result, &result); err != nil {
+ t.Fatalf("decode port result: %v", err)
+ }
+ if result.Port != target || h.proxyRebinds.Load() != 1 {
+ t.Fatalf("port=%d rebinds=%d, want port=%d and one rebind", result.Port, h.proxyRebinds.Load(), target)
+ }
+ ready, actual := h.b.getProxy().Status("llamacpp")
+ if !ready || actual != target {
+ t.Fatalf("llama.cpp facade ready=%v port=%d, want %d", ready, actual, target)
+ }
+ for engine, want := range otherPorts {
+ _, actual := h.b.getProxy().Status(engine)
+ if actual != want {
+ t.Fatalf("%s facade moved from %d to %d", engine, want, actual)
+ }
+ }
+ path, err := h.b.engineSettingsPath()
+ if err != nil {
+ t.Fatalf("resolve journal path: %v", err)
+ }
+ data, err := os.ReadFile(path)
+ if err != nil {
+ t.Fatalf("read journal: %v", err)
+ }
+ var records map[string]*engineSettingsRecord
+ if err := json.Unmarshal(data, &records); err != nil {
+ t.Fatalf("decode journal: %v", err)
+ }
+ record := records["llamacpp"]
+ want := before.Settings
+ want.ProxyPort = target
+ if record == nil || !record.Explicit || record.Snapshot.Settings != want || record.Snapshot.Phase != "succeeded" || record.Snapshot.Revision != before.Revision+1 {
+ t.Fatalf("persisted record = %+v, want successful explicit proxy-port change", record)
+ }
+}
diff --git a/services/nvpair-workload-manager/spec.md b/services/nvpair-workload-manager/spec.md
index 0c0b4d57..6e75ccfc 100644
--- a/services/nvpair-workload-manager/spec.md
+++ b/services/nvpair-workload-manager/spec.md
@@ -48,7 +48,7 @@ Tracks inference workloads cluster-wide as they are queued, executed, and retire
- For each validated, deduplicated inter-node `workload:*`, emit `workloads:upsert` to the Broker (translated, not forwarded unchanged); for each inter-node `workloads:remove`, emit `workloads:remove` to the Broker (not re-broadcast).
- Record an inter-node dedup key only after the corresponding notification is successfully written to the Broker. If the write fails, return `500` and leave the key available for a retry; concurrent requests for the same key must not both emit successfully.
- Subscribe to the Broker's discovery relay with `discovery:subscribe` filtered to the `wl` service, and rebuild the broadcast target set from every `discovery:nodes` snapshot: take each peer's dialable address and `wl` port from its directory entry, skip entries advertising no `wl` port or no address, and exclude this node's own entry by `hostUuid` rather than by hostname. A snapshot carries the full filtered set and replaces the target set wholesale, so there are no per-node deltas to apply and a peer absent from a snapshot simply stops being a target.
-- Deduplicate inbound lifecycle events by `(nodeId, engine, runId, Workload.id, state, scheduledOn, seq)` and removals by `(nodeId, workloadId)`, using a configurable bounded LRU index (default ~10,000 entries, sized for session-scoped volume at ~dozen-node scale). `nodeId` is part of the key because `Workload.id` is only unique per node (§11) — keying on `id` alone would collide across nodes and silently drop a legitimate peer's event. `engine` and `runId` are there for the same reason one level down: `Workload.id` is a per-process counter, both engine proxies count from 1, and the counter resets on restart, so without them a concurrent Ollama and LM Studio job both holding id `"1"` would collapse into one. The client-visible identity the Broker uses as the global key remains the coarser `(nodeId, workloadId)` pair (§10); this key is finer on purpose.
+- Deduplicate inbound lifecycle events by `(nodeId, engine, runId, Workload.id, state, scheduledOn, seq)` and removals by `(nodeId, workloadId)`, using a configurable bounded LRU index (default ~10,000 entries, sized for session-scoped volume at ~dozen-node scale). `nodeId` is part of the key because `Workload.id` is only unique per node (§11) — keying on `id` alone would collide across nodes and silently drop a legitimate peer's event. `engine` and `runId` are there for the same reason one level down: `Workload.id` is a per-process counter, every engine facade counts from 1, and the counter resets on restart, so without them concurrent jobs from different engines both holding id `"1"` would collapse into one. The client-visible identity the Broker uses as the global key remains the coarser `(nodeId, workloadId)` pair (§10); this key is finer on purpose.
- Validate inbound payloads; reject malformed envelopes or unknown `method` values (`400 Bad Request`).
**Non-functional**
diff --git a/services/readme.md b/services/readme.md
index 4ce9b5d9..a880ed57 100644
--- a/services/readme.md
+++ b/services/readme.md
@@ -11,9 +11,10 @@ local network: each node advertises itself over mDNS as one consolidated
node offers and where to reach them.
What a discovered node can actually serve is a separate question, answered after
-discovery. A node may be running [Ollama](https://ollama.com/), LM Studio, both,
-or neither, and its model inventory is fetched over HTTP from its engine-manager
-rather than crammed into mDNS TXT records, which are too small to carry it.
+discovery. A node may be running [Ollama](https://ollama.com/), LM Studio,
+llama.cpp, any combination, or none, and its model inventory is fetched over
+HTTP from its engine-manager rather than crammed into mDNS TXT records, which
+are too small to carry it.
Locally, each node exposes compatibility proxies — Ollama-compatible and
OpenAI-compatible — so an unmodified client on that machine can reach any capable
@@ -27,6 +28,13 @@ can serve that loopback-only address from the same router when it is free —
`localhost` is claimed on IPv4 and IPv6 together, and remote or HTTPS targets are
never intercepted.
+The llama.cpp facade is enabled by default alongside Ollama and LM Studio. It
+exposes OpenAI-compatible traffic on `http://localhost:8080` while the managed
+router runs on `8081`; a safe fallback is reported when the facade port is
+occupied. The desktop and TUI expose its managed lifecycle, downloads,
+inventory, and load/unload actions. `--proxy-engines` can still restrict the
+facades a standalone broker or TUI starts.
+
PAIR ships a graphical UI alongside these services. The UI launches
**`nvpair-ui-broker`** from the same directory; the broker orchestrates the
workers and exposes a newline-delimited JSON-RPC 2.0 API over stdio (or a Unix
@@ -62,10 +70,11 @@ configuration.
The broker feeds every accepted local or peer workload transition plus compact
GPU telemetry to the scheduler. Queued and running work is counted by destination
-node across Ollama and LM Studio together. Fresh maximum-GPU utilization is
-smoothed into pressure 0–3; missing or stale telemetry is neutral. Rankings use
-`pending + gpuPressure`, and each proxy adds local reservations before choosing,
-so bursts spread without waiting for workload feedback.
+node across Ollama, LM Studio, and llama.cpp together. Fresh maximum-GPU
+utilization is smoothed into pressure 0–3; missing or stale telemetry is
+neutral. Rankings use `pending + gpuPressure`, and each facade adds local
+reservations before choosing, so bursts spread without waiting for workload
+feedback.
## Repository layout
diff --git a/services/shared/engines/engines.go b/services/shared/engines/engines.go
index b7f1e8c0..28b4a94a 100644
--- a/services/shared/engines/engines.go
+++ b/services/shared/engines/engines.go
@@ -92,8 +92,8 @@ type Engine struct {
// claims in managed mode.
FacadePort int
- // EnginePortBase is where PAIR relocates the engine so the proxy can take
- // FacadePort, and the base of the next-free-port search.
+ // EnginePortBase is where PAIR runs or relocates the engine so the proxy can
+ // take FacadePort, and the base of any next-free-port search.
EnginePortBase int
// PortFile is the per-user file this engine's proxy persists its chosen
@@ -143,6 +143,14 @@ var all = []Engine{
EnginePortBase: 1235,
PortFile: "lmstudio-proxy-port.json",
},
+ {
+ Name: "llamacpp",
+ DisplayName: "llama.cpp",
+ DiscoveryService: noderec.ServiceLlamaCPP,
+ FacadePort: 8080,
+ EnginePortBase: 8081,
+ PortFile: "llamacpp-proxy-port.json",
+ },
}
// All returns the engine set in preparation order. The result is a copy, so a
diff --git a/services/shared/engines/engines_test.go b/services/shared/engines/engines_test.go
index aa609be8..1cc0af55 100644
--- a/services/shared/engines/engines_test.go
+++ b/services/shared/engines/engines_test.go
@@ -23,7 +23,7 @@ func TestOllamaIsPreparedFirst(t *testing.T) {
func TestNames(t *testing.T) {
got := Names()
- want := []string{"ollama", "lmstudio"}
+ want := []string{"ollama", "lmstudio", "llamacpp"}
if len(got) != len(want) {
t.Fatalf("Names() = %v, want %v", got, want)
}
diff --git a/services/shared/noderec/noderec.go b/services/shared/noderec/noderec.go
index 112e7fe3..f44a263d 100644
--- a/services/shared/noderec/noderec.go
+++ b/services/shared/noderec/noderec.go
@@ -10,7 +10,7 @@
// whose TXT map carries a schema version, the node's identity, its LAN address,
// and one compact key per local service port, e.g.:
//
-// v=1;uuid=;cluster-uuid=;ip=192.168.1.10;ni=14318;ol=11434;lm=1234;er=14319;wl=14320;cl=14321;em=14322
+// v=1;uuid=;cluster-uuid=;ip=192.168.1.10;ni=14318;ol=11434;lm=1234;lc=8080;er=14319;wl=14320;cl=14321;em=14322
//
// Design decisions this package encodes:
// - SRV port is a fixed, NON-authoritative constant; consumers ignore it and
@@ -87,6 +87,7 @@ const (
ServiceNodeInfo ServiceKey = "ni"
ServiceOllama ServiceKey = "ol"
ServiceLMStudio ServiceKey = "lm"
+ ServiceLlamaCPP ServiceKey = "lc"
ServiceErrors ServiceKey = "er"
ServiceWorkload ServiceKey = "wl"
ServiceCluster ServiceKey = "cl"
@@ -104,7 +105,7 @@ const (
// serviceKeyOrder is the deterministic emit order for service ports in TXT.
var serviceKeyOrder = []ServiceKey{
- ServiceNodeInfo, ServiceOllama, ServiceLMStudio,
+ ServiceNodeInfo, ServiceOllama, ServiceLMStudio, ServiceLlamaCPP,
ServiceErrors, ServiceWorkload, ServiceCluster, ServiceEngineManager,
ServiceEngineControl,
}
diff --git a/services/shared/noderec/noderec_test.go b/services/shared/noderec/noderec_test.go
index 8b0757e5..6a7bc1bf 100644
--- a/services/shared/noderec/noderec_test.go
+++ b/services/shared/noderec/noderec_test.go
@@ -147,6 +147,7 @@ func TestTransportPolicy(t *testing.T) {
{ServiceNodeInfo, TransportPlain, false, false},
{ServiceOllama, TransportPlain, false, false},
{ServiceLMStudio, TransportPlain, false, false},
+ {ServiceLlamaCPP, TransportPlain, false, false},
{ServiceEngineManager, TransportPlain, false, false},
{ServiceErrors, TransportMTLSWhenClustered, true, false},
{ServiceWorkload, TransportMTLSWhenClustered, true, false},
diff --git a/services/tests/broker_supervision_test.go b/services/tests/broker_supervision_test.go
index 87a9fce0..98613274 100644
--- a/services/tests/broker_supervision_test.go
+++ b/services/tests/broker_supervision_test.go
@@ -276,22 +276,22 @@ func mustPid(t *testing.T, line string) int {
return pid
}
-// Both engines are fronted by ONE process. This is the property the whole
+// Every default engine is fronted by ONE process. This is the property the whole
// unification exists for: the estimated-work reservations a proxy takes
-// between scheduler snapshots live in the process, so two processes each held
-// half the picture and could dispatch simultaneous bursts to the same node
-// each believing it idle.
+// between scheduler snapshots live in the process, so separate processes each
+// held an incomplete picture and could dispatch simultaneous bursts to the
+// same node while each believed it idle.
//
// Asserted on pids rather than on the code shape, because "one supervisor" is
-// an implementation detail and "one OS process serving both engines" is the
+// an implementation detail and "one OS process serving every engine" is the
// thing that actually makes the reservation map whole.
-func TestBothEnginesAreServedByOneProxyProcess(t *testing.T) {
+func TestAllDefaultEnginesAreServedByOneProxyProcess(t *testing.T) {
if portBusy(11435) || portBusy(1234) {
t.Skip("ollama-proxy (11435) or lmstudio-proxy (1234) default port already in use; skipping")
}
stdin, msgs, stderr, cleanup := startBrokerWith(t,
- // No --proxy-engines: both engines, which is the default.
+ // No --proxy-engines: exercise the complete default set.
"--proxy-path", proxyBin,
)
t.Cleanup(cleanup)
@@ -309,14 +309,17 @@ func TestBothEnginesAreServedByOneProxyProcess(t *testing.T) {
waitForMethod(t, msgs, "app:ready", 10*time.Second)
- // Both facades must actually come up, or "one process" would be trivially
+ // Every facade must actually come up, or "one process" would be trivially
// true by one of them having failed.
ollamaPort := waitProxyReady(t, stdin, msgs, 15*time.Second)
lmstudioPort := waitLMStudioProxyReady(t, stdin, msgs, 15*time.Second)
- if ollamaPort == lmstudioPort {
- t.Fatalf("both facades report port %d; they must bind separately", ollamaPort)
+ llamacppPort := waitEngineProxyReady(t, "llamacpp-proxy", stdin, msgs, 15*time.Second)
+ if len(map[int]bool{ollamaPort: true, lmstudioPort: true, llamacppPort: true}) != 3 {
+ t.Fatalf("facade ports must be distinct: ollama=%d lmstudio=%d llamacpp=%d",
+ ollamaPort, lmstudioPort, llamacppPort)
}
- t.Logf("ollama facade on :%d, lmstudio facade on :%d", ollamaPort, lmstudioPort)
+ t.Logf("facades ready: ollama=%d lmstudio=%d llamacpp=%d",
+ ollamaPort, lmstudioPort, llamacppPort)
// Collect every spawn announced up to this point plus a settling window, so
// a second process starting late still shows up.
@@ -331,9 +334,9 @@ func TestBothEnginesAreServedByOneProxyProcess(t *testing.T) {
}
}
if len(seen) != 1 {
- t.Fatalf("saw %d proxy processes %v, want exactly 1 hosting both engines", len(seen), seen)
+ t.Fatalf("saw %d proxy processes %v, want exactly 1 hosting all default engines", len(seen), seen)
}
- t.Logf("both engines served by pid %v", seen)
+ t.Logf("all default engines served by pid %v", seen)
}
// TestBrokerShutsDownOnSignal verifies the broker exits promptly when it
diff --git a/services/tests/llamacpp_interop_test.go b/services/tests/llamacpp_interop_test.go
new file mode 100644
index 00000000..46684834
--- /dev/null
+++ b/services/tests/llamacpp_interop_test.go
@@ -0,0 +1,171 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+package tests
+
+import (
+ "bytes"
+ "encoding/json"
+ "fmt"
+ "io"
+ "net/http"
+ "net/http/httptest"
+ "sync/atomic"
+ "testing"
+ "time"
+)
+
+func TestLlamaCPPProxyIsIncludedInBrokerDefaults(t *testing.T) {
+ stdin, msgs, stderr, cleanup := startBrokerWith(t, "--proxy-path", proxyBin)
+ t.Cleanup(cleanup)
+ go func() {
+ for range stderr {
+ }
+ }()
+
+ waitForMethod(t, msgs, "app:ready", 10*time.Second)
+ if port := waitEngineProxyReady(t, "llamacpp-proxy", stdin, msgs, 15*time.Second); port <= 0 {
+ t.Fatalf("default llama.cpp proxy port = %d, want a listening facade", port)
+ }
+}
+
+func TestBrokerLlamaCPPProxySetPortRejectsInvalidPorts(t *testing.T) {
+ stdin, msgs, stderr, cleanup := startBrokerWith(t,
+ "--proxy-path", proxyBin, "--proxy-engines", "llamacpp",
+ )
+ t.Cleanup(cleanup)
+ go func() {
+ for range stderr {
+ }
+ }()
+ waitForMethod(t, msgs, "app:ready", 10*time.Second)
+ for i, tc := range []struct{ name, params string }{
+ {"missing port", `{}`},
+ {"zero port", `{"port":0}`},
+ {"negative port", `{"port":-1}`},
+ {"oversized port", `{"port":65536}`},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ id := 7300 + i
+ if _, err := fmt.Fprintf(stdin, `{"jsonrpc":"2.0","id":%d,"method":"llamacpp-proxy:set-port","params":%s}`+"\n", id, tc.params); err != nil {
+ t.Fatalf("write proxy port request: %v", err)
+ }
+ response := waitForResponse(t, msgs, 10*time.Second)
+ if response.ID == nil || string(*response.ID) != fmt.Sprint(id) {
+ t.Fatalf("unexpected response ID: %+v", response)
+ }
+ if response.Error == nil || response.Error.Code != -32602 || response.Error.Message != "port must be between 1 and 65535" {
+ t.Fatalf("error = %+v, want broker invalid-port rejection", response.Error)
+ }
+ })
+ }
+}
+
+func TestLlamaCPPFacadeUsesRouterInventoryAndExactModelIDs(t *testing.T) {
+ const model = "org/router-model-GGUF:Q4_K_M"
+ var modelListHits atomic.Int32
+ var inferenceHits atomic.Int32
+ upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ w.Header().Set("Content-Type", "application/json")
+ switch {
+ case r.Method == http.MethodGet && r.URL.Path == "/models":
+ modelListHits.Add(1)
+ _ = json.NewEncoder(w).Encode(map[string]any{
+ "object": "list",
+ "data": []map[string]string{{"id": model}},
+ })
+ case r.Method == http.MethodPost && r.URL.Path == "/v1/chat/completions":
+ inferenceHits.Add(1)
+ _, _ = io.Copy(io.Discard, r.Body)
+ _, _ = io.WriteString(w, `{"choices":[]}`)
+ default:
+ http.NotFound(w, r)
+ }
+ }))
+ t.Cleanup(upstream.Close)
+
+ stdin, msgs, stderr, cleanup := startBrokerWith(t,
+ "--proxy-path", proxyBin, "--proxy-engines", "llamacpp",
+ )
+ t.Cleanup(cleanup)
+ go func() {
+ for range stderr {
+ }
+ }()
+
+ waitForMethod(t, msgs, "app:ready", 10*time.Second)
+ proxyPort := waitEngineProxyReady(t, "llamacpp-proxy", stdin, msgs, 15*time.Second)
+ callBrokerRPC(t, stdin, msgs, 7200, "llamacpp-proxy:node/add-manual", map[string]any{
+ "id": "llamacpp-owner",
+ "host": "127.0.0.1",
+ "port": portOfURL(t, upstream.URL),
+ "addresses": []string{"127.0.0.1"},
+ "models": []string{model},
+ })
+
+ client := &http.Client{Timeout: 5 * time.Second}
+ t.Cleanup(client.CloseIdleConnections)
+ getModelList := func(t *testing.T, path string) {
+ t.Helper()
+ hitsBefore := modelListHits.Load()
+ response, err := client.Get(fmt.Sprintf("http://127.0.0.1:%d%s", proxyPort, path))
+ if err != nil {
+ t.Fatalf("get model list: %v", err)
+ }
+ defer response.Body.Close()
+ var list struct {
+ Object string `json:"object"`
+ Data []struct {
+ ID string `json:"id"`
+ } `json:"data"`
+ }
+ if err := json.NewDecoder(response.Body).Decode(&list); err != nil {
+ t.Fatalf("decode model list: %v", err)
+ }
+ found := false
+ for _, item := range list.Data {
+ if item.ID == model {
+ found = true
+ }
+ }
+ if response.StatusCode != http.StatusOK || list.Object != "list" ||
+ !found || modelListHits.Load() != hitsBefore+1 {
+ t.Fatalf("model list status=%d body=%+v upstreamHits=%d",
+ response.StatusCode, list, modelListHits.Load())
+ }
+ }
+ t.Run("remaps OpenAI model list to router inventory", func(t *testing.T) {
+ getModelList(t, "/v1/models")
+ })
+ t.Run("serves router model-list alias", func(t *testing.T) {
+ getModelList(t, "/models")
+ })
+
+ post := func(t *testing.T, requestedModel string) int {
+ t.Helper()
+ endpoint := fmt.Sprintf("http://127.0.0.1:%d/v1/chat/completions", proxyPort)
+ response, err := client.Post(endpoint, "application/json",
+ bytes.NewBufferString(fmt.Sprintf(`{"model":%q,"messages":[]}`, requestedModel)))
+ if err != nil {
+ t.Fatalf("post inference: %v", err)
+ }
+ if _, err := io.Copy(io.Discard, response.Body); err != nil {
+ t.Fatalf("read inference response: %v", err)
+ }
+ if err := response.Body.Close(); err != nil {
+ t.Fatalf("close inference response: %v", err)
+ }
+ return response.StatusCode
+ }
+ t.Run("routes only the exact advertised model id", func(t *testing.T) {
+ if status := post(t, model); status != http.StatusOK {
+ t.Fatalf("matching model status = %d, want 200", status)
+ }
+ if status := post(t, "org/router-model-GGUF:q4_k_m"); status != http.StatusBadGateway {
+ t.Fatalf("case-changed model status = %d, want 502", status)
+ }
+ if got := inferenceHits.Load(); got != 1 {
+ t.Fatalf("upstream inference hits = %d, want only the exact match", got)
+ }
+ })
+}
diff --git a/services/tests/model_routing_interop_test.go b/services/tests/model_routing_interop_test.go
index 58ef22ea..58056042 100644
--- a/services/tests/model_routing_interop_test.go
+++ b/services/tests/model_routing_interop_test.go
@@ -48,9 +48,7 @@ func TestStrictModelRoutingAcrossProcesses(t *testing.T) {
ineligible := newRoutingUpstream(t, http.StatusOK)
stdin, msgs, stderr, cleanup := startBrokerWith(t,
- // No --proxy-engines: this test wants both engines fronted, which is
- // the default.
- "--proxy-path", proxyBin,
+ "--proxy-path", proxyBin, "--proxy-engines", "ollama,lmstudio",
)
t.Cleanup(cleanup)
go func() {
diff --git a/services/tests/models_http_test.go b/services/tests/models_http_test.go
index baf5601d..4a49af10 100644
--- a/services/tests/models_http_test.go
+++ b/services/tests/models_http_test.go
@@ -33,14 +33,16 @@ func TestModelsHTTPEnrichment(t *testing.T) {
// Flat union + per-engine attribution, exactly the shape engine-manager's
// ModelsResult serializes. The daemon must enrich both onto the node.
_ = json.NewEncoder(w).Encode(map[string]any{
- "models": []string{"llama3:8b", "qwen:0.5b"},
+ "models": []string{"llama3:8b", "qwen:0.5b", "org/router-model-GGUF:Q4_K_M"},
"modelsByEngine": map[string][]string{
"ollama": {"llama3:8b"},
"lmstudio": {"qwen:0.5b"},
+ "llamacpp": {"org/router-model-GGUF:Q4_K_M"},
},
"loadedByEngine": map[string][]string{
"ollama": {"llama3:8b"},
"lmstudio": {},
+ "llamacpp": {"org/router-model-GGUF:Q4_K_M"},
},
})
}))
@@ -110,7 +112,7 @@ func findNode(nodes []availableNode, id string) (availableNode, bool) {
}
func modelsMatch(models []string) bool {
- want := []string{"llama3:8b", "qwen:0.5b"}
+ want := []string{"llama3:8b", "qwen:0.5b", "org/router-model-GGUF:Q4_K_M"}
if len(models) != len(want) {
return false
}
@@ -126,17 +128,19 @@ func modelsByEngineMatch(byEngine map[string][]string) bool {
want := map[string][]string{
"ollama": {"llama3:8b"},
"lmstudio": {"qwen:0.5b"},
+ "llamacpp": {"org/router-model-GGUF:Q4_K_M"},
}
return byEngineEqual(byEngine, want)
}
-// loadedByEngineMatch asserts the loaded set the stub served (ollama has one
-// resident model; lmstudio is running but empty) survives the daemon->broker
-// projection onto the client-facing node.
+// loadedByEngineMatch asserts the loaded set the stub served (Ollama and
+// llama.cpp each have one resident model; LM Studio is running but empty)
+// survives the daemon->broker projection onto the client-facing node.
func loadedByEngineMatch(loaded map[string][]string) bool {
want := map[string][]string{
"ollama": {"llama3:8b"},
"lmstudio": {},
+ "llamacpp": {"org/router-model-GGUF:Q4_K_M"},
}
return byEngineEqual(loaded, want)
}
diff --git a/services/tests/models_refresh_test.go b/services/tests/models_refresh_test.go
index 60878798..d04039ce 100644
--- a/services/tests/models_refresh_test.go
+++ b/services/tests/models_refresh_test.go
@@ -93,14 +93,16 @@ func TestModelsPeriodicRefreshConvergesWithoutMDNSChange(t *testing.T) {
// only the periodic refresh loop can converge the directory. The deadline
// comfortably exceeds the refresh interval so at least one sweep runs.
stub.set(map[string]any{
- "models": []string{"llama3:8b", "qwen:0.5b"},
+ "models": []string{"llama3:8b", "qwen:0.5b", "org/router-model-GGUF:Q4_K_M"},
"modelsByEngine": map[string][]string{
"ollama": {"llama3:8b"},
"lmstudio": {"qwen:0.5b"},
+ "llamacpp": {"org/router-model-GGUF:Q4_K_M"},
},
"loadedByEngine": map[string][]string{
"ollama": {"llama3:8b"},
"lmstudio": {},
+ "llamacpp": {"org/router-model-GGUF:Q4_K_M"},
},
})
pollForNode(t, stdin, msgs, instance, 45*time.Second, func(n availableNode) bool {
@@ -116,13 +118,15 @@ func TestModelsPeriodicRefreshConvergesWithoutMDNSChange(t *testing.T) {
"modelsByEngine": map[string][]string{
"ollama": {},
"lmstudio": {},
+ "llamacpp": {},
},
"loadedByEngine": map[string][]string{
"ollama": {},
"lmstudio": {},
+ "llamacpp": {},
},
})
- emptyByEngine := map[string][]string{"ollama": {}, "lmstudio": {}}
+ emptyByEngine := map[string][]string{"ollama": {}, "lmstudio": {}, "llamacpp": {}}
pollForNode(t, stdin, msgs, instance, 45*time.Second, func(n availableNode) bool {
return len(n.Models) == 0 && byEngineEqual(n.ModelsByEngine, emptyByEngine) &&
byEngineEqual(n.LoadedByEngine, emptyByEngine)
diff --git a/services/tests/workload_identity_interop_test.go b/services/tests/workload_identity_interop_test.go
index 864a3e58..2300c356 100644
--- a/services/tests/workload_identity_interop_test.go
+++ b/services/tests/workload_identity_interop_test.go
@@ -66,8 +66,7 @@ func TestWorkloadCrossEngineIdentityDistinct(t *testing.T) {
lmstudioPort := portOfURL(t, lmstudio.URL)
stdin, msgs, _, cleanup := startBrokerWith(t,
- // Both engines fronted, which is the default.
- "--proxy-path", proxyBin,
+ "--proxy-path", proxyBin, "--proxy-engines", "ollama,lmstudio",
"--workload-manager-path", workloadMgrBin,
)
t.Cleanup(cleanup)