Skip to content

Commit 2eadb00

Browse files
committed
Probe manual nodes for llmman on 17434 as an Ollama-API engine
llmman (https://github.com/llmmanorg/llmman) serves the Ollama API on 17434, but manual nodes were only probed for Ollama on 11434, so a node running llmman showed no inference engine. Retry the probe on 17434 and report the answering port in ollama_port; the broker and ollama-proxy already honour that field, so nothing else changes. Ollama wins when both answer. Signed-off-by: Eric Curtin <eric.curtin@docker.com>
1 parent bdd010c commit 2eadb00

3 files changed

Lines changed: 62 additions & 12 deletions

File tree

‎services/nvpair-manual-nodes/README.md‎

Lines changed: 4 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -62,6 +62,8 @@ Emitted when a manually added node has been probed and its initial status determ
6262

6363
Each node is probed for both inference engines: Ollama on its default `:11434` (`GET /` + `/api/tags`) and LM Studio on its default `:1234` (`GET /v1/models`, which doubles as the liveness check and the model list). `lmstudio_up` / `lmstudio_port` / `lmstudio_models` mirror the `ollama_*` fields and let a supervising broker bridge the node into that engine's `nvpair-proxy` instance the same way it bridges Ollama into its own. A node can run either engine, both, or neither.
6464

65+
When nothing answers on `:11434`, the same Ollama probe is retried on `:17434`, the default port of [llmman](https://github.com/llmmanorg/llmman), which serves the Ollama API. Such a node reports `ollama_up: true` with `ollama_port: 17434` and is bridged into `ollama-proxy` like an Ollama node.
66+
6567
### `node/updated`
6668

6769
Emitted when a periodic probe detects a change (service going up/down, models,
@@ -134,13 +136,13 @@ Changes the log level at runtime. Accepted as either a request (answered with `{
134136

135137
Each manual node is probed every 10 seconds, with a 3-second timeout per leg, for:
136138

137-
- **Ollama** on port 11434: health check (`GET /`) and model list (`GET /api/tags`)
139+
- **Ollama** on port 11434: health check (`GET /`) and model list (`GET /api/tags`). If it does not answer, the same probe is sent to port 17434, where [llmman](https://github.com/llmmanorg/llmman) serves the Ollama API; `ollama_port` reports whichever port answered
138140
- **LM Studio** on port 1234: `GET /v1/models`, which doubles as the liveness check and the model list
139141
- **Node Info** on port 14318, or `tls_port` over HTTPS: hardware inventory and identity (`GET /v1/node-info`)
140142

141143
A node can have any combination of these, or none if the target is unreachable. Status changes trigger `node/updated` events. Because change detection compares CPU, memory, and GPU values, a node running node-info emits a `node/updated` on most probe cycles as utilization moves.
142144

143-
The three engine ports are compiled in: only the node-info leg's port can be moved, via `tls_port`. A remote engine on a non-default port is not discovered.
145+
The engine ports are compiled in: only the node-info leg's port can be moved, via `tls_port`. A remote engine on a non-default port is not discovered.
144146

145147
## Shutdown
146148

‎services/nvpair-manual-nodes/manager.go‎

Lines changed: 21 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -251,7 +251,7 @@ func (m *Manager) probeNode(entry ManualEntry) {
251251
addr := entry.Address
252252
id := nodeID(entry)
253253

254-
ollamaUp, ollamaModels := m.probeOllama(addr, 11434)
254+
ollamaUp, ollamaAPIPort, ollamaModels := m.probeOllamaAPI(addr)
255255
lmStudioUp, lmStudioModels := m.probeLMStudio(addr, lmStudioPort)
256256

257257
// Pick scheme + port + client based on the entry's TLS hint.
@@ -285,7 +285,7 @@ func (m *Manager) probeNode(entry ManualEntry) {
285285
Name: entry.Name,
286286
Address: addr,
287287
OllamaUp: ollamaUp,
288-
OllamaPort: 11434,
288+
OllamaPort: ollamaAPIPort,
289289
OllamaModels: ollamaModels,
290290
LMStudioUp: lmStudioUp,
291291
LMStudioPort: lmStudioPort,
@@ -403,6 +403,24 @@ func probeFailedID(nodeID string) string {
403403
// manager (which only governs the local engine).
404404
const lmStudioPort = 1234
405405

406+
// Default ports of Ollama and of llmman (https://github.com/llmmanorg/llmman),
407+
// which serves the same Ollama API on 17434.
408+
const (
409+
ollamaPort = 11434
410+
llmmanPort = 17434
411+
)
412+
413+
// probeOllamaAPI probes Ollama first, then llmman, and returns whether one
414+
// answered, its port (ollamaPort when neither did) and its models.
415+
func (m *Manager) probeOllamaAPI(addr string) (bool, int, []string) {
416+
for _, port := range []int{ollamaPort, llmmanPort} {
417+
if up, models := m.probeOllama(addr, port); up {
418+
return true, port, models
419+
}
420+
}
421+
return false, ollamaPort, nil
422+
}
423+
406424
// probeLMStudio checks LM Studio's OpenAI-compatible server on addr:port. A
407425
// single GET /v1/models doubles as the liveness check and the model list (the
408426
// response is {"data":[{"id":"..."}],...}). Returns whether it is up and the
@@ -539,7 +557,7 @@ func (m *Manager) addNode(entry ManualEntry) ManualNodeStatus {
539557
ID: id,
540558
Name: entry.Name,
541559
Address: entry.Address,
542-
OllamaPort: 11434,
560+
OllamaPort: ollamaPort,
543561
NodeInfoPort: nodeInfoPort,
544562
TLSEnabled: entry.TLSPort > 0,
545563
MTLSRequired: entry.TLSPort > 0 && entry.MTLS,

‎services/nvpair-manual-nodes/manager_test.go‎

Lines changed: 37 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -11,6 +11,7 @@ import (
1111
"io"
1212
"net"
1313
"net/http"
14+
"strconv"
1415
"strings"
1516
"sync"
1617
"testing"
@@ -84,7 +85,18 @@ func newTestManager() (*Manager, *captureRW, *fakeRoundTripper) {
8485
}
8586

8687
func configureHealthyNode(rt *fakeRoundTripper, addr string, models []string, info NodeInfoResponse) {
87-
host := net.JoinHostPort(addr, "11434")
88+
configureHealthyOllamaAPI(rt, addr, ollamaPort, models)
89+
90+
infoHost := net.JoinHostPort(addr, "14318")
91+
rt.set(http.MethodGet, infoHost, "/v1/node-info", func(*http.Request) (*http.Response, error) {
92+
data, _ := json.Marshal(info)
93+
return httpJSON(http.StatusOK, string(data))
94+
})
95+
}
96+
97+
// configureHealthyOllamaAPI registers GET / and GET /api/tags on addr:port.
98+
func configureHealthyOllamaAPI(rt *fakeRoundTripper, addr string, port int, models []string) {
99+
host := net.JoinHostPort(addr, strconv.Itoa(port))
88100
rt.set(http.MethodGet, host, "/", func(*http.Request) (*http.Response, error) {
89101
return httpJSON(http.StatusOK, `{}`)
90102
})
@@ -102,12 +114,6 @@ func configureHealthyNode(rt *fakeRoundTripper, addr string, models []string, in
102114
data, _ := json.Marshal(payload)
103115
return httpJSON(http.StatusOK, string(data))
104116
})
105-
106-
infoHost := net.JoinHostPort(addr, "14318")
107-
rt.set(http.MethodGet, infoHost, "/v1/node-info", func(*http.Request) (*http.Response, error) {
108-
data, _ := json.Marshal(info)
109-
return httpJSON(http.StatusOK, string(data))
110-
})
111117
}
112118

113119
// configureHealthyLMStudio registers a 200 GET /v1/models on addr:1234
@@ -154,6 +160,30 @@ func TestProbeLMStudioReportsModels(t *testing.T) {
154160
}
155161
}
156162

163+
// TestProbeOllamaAPIFallsBackToLLMMan: Ollama wins when both answer, llmman is
164+
// reported with its port when only it answers, neither reports down on 11434.
165+
func TestProbeOllamaAPIFallsBackToLLMMan(t *testing.T) {
166+
m, _, rt := newTestManager()
167+
configureHealthyOllamaAPI(rt, "both.local", ollamaPort, []string{"llama3"})
168+
configureHealthyOllamaAPI(rt, "both.local", llmmanPort, []string{"gemma4"})
169+
configureHealthyOllamaAPI(rt, "llmman.local", llmmanPort, []string{"gemma4", "qwen3.8"})
170+
171+
up, port, models := m.probeOllamaAPI("both.local")
172+
if !up || port != ollamaPort || len(models) != 1 || models[0] != "llama3" {
173+
t.Fatalf("both: up=%v port=%d models=%#v, want ollama on %d", up, port, models, ollamaPort)
174+
}
175+
176+
up, port, models = m.probeOllamaAPI("llmman.local")
177+
if !up || port != llmmanPort || len(models) != 2 || models[0] != "gemma4" || models[1] != "qwen3.8" {
178+
t.Fatalf("llmman: up=%v port=%d models=%#v, want llmman on %d", up, port, models, llmmanPort)
179+
}
180+
181+
up, port, models = m.probeOllamaAPI("absent.local")
182+
if up || port != ollamaPort || models != nil {
183+
t.Fatalf("absent: up=%v port=%d models=%#v, want down on %d", up, port, models, ollamaPort)
184+
}
185+
}
186+
157187
func requestMessage(id int, method string, params any) *Message {
158188
idData, _ := json.Marshal(id)
159189
idRaw := json.RawMessage(idData)

0 commit comments

Comments
 (0)