Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
26 commits
Select commit Hold shift + click to select a range
edef785
Refresh deprecating NIMs: LLM->3.5-lightning, embedder->nemotron-3-em…
karthikt-nvidia Aug 19, 2026
56afd5b
post-purchase: force json_object so multi-line message stays valid JSON
karthikt-nvidia Aug 19, 2026
3263988
recommendation: force json_object across ARAG pipeline stages
karthikt-nvidia Aug 19, 2026
454db77
deploy: align NIM model replacements
antoniomtz Aug 19, 2026
b8c86cd
deploy: enable Lightning automatic tool calling
antoniomtz Aug 19, 2026
7139866
search: recalibrate relevance cutoffs for nemotron-3-embed-1b embedde…
nhuphuc1602 Aug 20, 2026
6c4c576
fix(apps-sdk): stabilize NIM widget flows
antoniomtz Aug 20, 2026
2b942e9
fix(apps-sdk): remove search tool-call timeout
antoniomtz Aug 20, 2026
eab340d
fix(apps-sdk): allow Lightning search completion
antoniomtz Aug 20, 2026
7d9dc9c
fix(ui): preserve Apps SDK tab pointer activation
antoniomtz Aug 20, 2026
6d550e1
fix(apps-sdk): preserve product card pointer activation
antoniomtz Aug 20, 2026
129ae7b
fix(apps-sdk): harden public inference workflows
antoniomtz Aug 20, 2026
51a1f8f
fix(apps-sdk): retry transient recommendation failures
antoniomtz Aug 20, 2026
59b90fd
fix(agents): bound public inference retries
antoniomtz Aug 20, 2026
89efc5e
fix(ci): retry transient NAT inference failures
antoniomtz Aug 20, 2026
dbc6ed7
fix(ci): back off transient inference retries
antoniomtz Aug 20, 2026
fc51525
fix(apps-sdk): coalesce hosted inference requests
antoniomtz Aug 20, 2026
0fa5120
test(ci): separate hosted inference workloads
antoniomtz Aug 20, 2026
36108d6
test(ci): retry transient empty recommendations
antoniomtz Aug 20, 2026
25f59b1
test(ci): pace hosted inference workflows
antoniomtz Aug 20, 2026
e1123ed
test(ci): isolate hosted inference rate windows
antoniomtz Aug 20, 2026
05dde23
test(ci): remove ineffective inference cooldowns
antoniomtz Aug 20, 2026
f5ffce8
test(ci): restore hosted inference cooldowns
antoniomtz Aug 21, 2026
5c6dcab
fix(agents): remove unsupported NIM retry parameter
antoniomtz Aug 21, 2026
fdaa374
[ci] Isolate transient Apps SDK inference failures (#131)
antoniomtz Aug 21, 2026
2d2a040
test(ci): retry transient NAT request timeouts
antoniomtz Aug 21, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
111 changes: 83 additions & 28 deletions .github/scripts/check_nat_agents.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,13 @@
EMPTY_LLM_RESPONSE_PREFIX = (
"LLM returned an empty response (no content, no tool calls)."
)
EMPTY_LLM_RESPONSE_RETRY_DELAY_SECONDS = 2
NO_AGENT_RESPONSE_PREFIX = "No response received from agent"
# The recommendation probe makes several genuine, parallel LLM calls immediately
# before the search probe. Give a shared hosted endpoint time to clear a short
# capacity window, then still require a real successful workflow response.
TRANSIENT_LLM_RESPONSE_RETRY_DELAYS_SECONDS = (2, 5, 10)
EMPTY_RECOMMENDATION_RETRY_DELAYS_SECONDS = (5, 10, 20)
WORKFLOW_COOLDOWN_SECONDS = 15


def require(condition: bool, message: str) -> None:
Expand Down Expand Up @@ -47,8 +53,8 @@ def require_object_list(
return objects


def is_retryable_empty_llm_response(status: int, detail: str) -> bool:
"""Return whether NAT reported the known transient empty-LLM failure."""
def is_retryable_transient_llm_response(status: int, detail: str) -> bool:
"""Return whether NAT reported a transient public-inference failure."""
if status != 422:
return False

Expand All @@ -65,7 +71,7 @@ def is_retryable_empty_llm_response(status: int, detail: str) -> bool:
payload.get("code") == "workflow_error"
and payload.get("details") == "RuntimeError"
and isinstance(message, str)
and message.startswith(EMPTY_LLM_RESPONSE_PREFIX)
and message.startswith((EMPTY_LLM_RESPONSE_PREFIX, NO_AGENT_RESPONSE_PREFIX))
)


Expand All @@ -76,7 +82,7 @@ def call_agent(
timeout: int,
) -> dict[str, Any]:
"""Execute a NAT workflow and unwrap its JSON response value."""
for attempt in range(2):
for attempt in range(len(TRANSIENT_LLM_RESPONSE_RETRY_DELAYS_SECONDS) + 1):
request = urllib.request.Request(
f"http://{agent}-agent:{port}/generate",
data=json.dumps({"input_message": json.dumps(input_message)}).encode(),
Expand All @@ -90,24 +96,41 @@ def call_agent(
raw_response = response.read().decode()
except urllib.error.HTTPError as error:
detail = error.read().decode(errors="replace")
if attempt == 0 and is_retryable_empty_llm_response(error.code, detail):
if attempt < len(
TRANSIENT_LLM_RESPONSE_RETRY_DELAYS_SECONDS
) and is_retryable_transient_llm_response(error.code, detail):
retry_delay = TRANSIENT_LLM_RESPONSE_RETRY_DELAYS_SECONDS[attempt]
print(
f"::warning::{agent} received a transient empty LLM response; "
"retrying once in 2 seconds.",
f"::warning::{agent} received a transient LLM response; "
f"retrying in {retry_delay} seconds.",
file=sys.stderr,
)
time.sleep(EMPTY_LLM_RESPONSE_RETRY_DELAY_SECONDS)
time.sleep(retry_delay)
continue
raise RuntimeError(
f"{agent} returned HTTP {error.code}: {detail[:1000]}"
) from error
except (TimeoutError, urllib.error.URLError) as error:
reason = getattr(error, "reason", str(error))
is_timeout = isinstance(error, TimeoutError) or isinstance(
reason, TimeoutError
)
if is_timeout and attempt < len(
TRANSIENT_LLM_RESPONSE_RETRY_DELAYS_SECONDS
):
retry_delay = TRANSIENT_LLM_RESPONSE_RETRY_DELAYS_SECONDS[attempt]
print(
f"::warning::{agent} request timed out; "
f"retrying in {retry_delay} seconds.",
file=sys.stderr,
)
time.sleep(retry_delay)
continue
raise RuntimeError(f"{agent} request failed: {reason}") from error

break
else:
raise RuntimeError(f"{agent} exhausted its empty-response retry")
raise RuntimeError(f"{agent} exhausted its transient-response retry")

require(status == 200, f"{agent} returned HTTP {status}")

Expand Down Expand Up @@ -210,24 +233,41 @@ def check_post_purchase() -> None:

def check_recommendation() -> None:
"""Verify Milvus-backed complementary product recommendations."""
result = call_agent(
"recommendation",
8004,
{
"query": "Recommend products that complement a Classic Tee",
"cart_items": [
{
"product_id": "prod_1",
"name": "Classic Tee",
"category": "tops",
"price": 2500,
}
],
"session_context": {"browse_history": ["casual wear", "jeans"]},
},
timeout=120,
)
recommendations = require_object_list(result, "recommendations", "recommendation")
request = {
"query": "Recommend products that complement a Classic Tee",
"cart_items": [
{
"product_id": "prod_1",
"name": "Classic Tee",
"category": "tops",
"price": 2500,
}
],
"session_context": {"browse_history": ["casual wear", "jeans"]},
}

for retry_delay in (*EMPTY_RECOMMENDATION_RETRY_DELAYS_SECONDS, None):
result = call_agent("recommendation", 8004, request, timeout=120)
raw_recommendations = result.get("recommendations")
if isinstance(raw_recommendations, list) and raw_recommendations:
recommendations = require_object_list(
result, "recommendations", "recommendation"
)
break

if retry_delay is None:
recommendations = require_object_list(
result, "recommendations", "recommendation"
)
break

print(
"::warning::recommendation returned no products after a live "
f"LLM workflow; retrying in {retry_delay} seconds.",
file=sys.stderr,
)
time.sleep(retry_delay)

require(
all(
bool(item.get("product_id")) and item.get("product_id") != "prod_1"
Expand Down Expand Up @@ -258,8 +298,23 @@ def main() -> int:
"""Run all NAT functional checks sequentially to avoid rate-limit bursts."""
try:
check_promotion()
print(
"::notice::waiting 15 seconds before the next live inference workflow.",
file=sys.stderr,
)
time.sleep(WORKFLOW_COOLDOWN_SECONDS)
check_post_purchase()
print(
"::notice::waiting 15 seconds before the parallel recommendation workflow.",
file=sys.stderr,
)
time.sleep(WORKFLOW_COOLDOWN_SECONDS)
check_recommendation()
print(
"::notice::waiting 15 seconds before the final search workflow.",
file=sys.stderr,
)
time.sleep(WORKFLOW_COOLDOWN_SECONDS)
check_search()
except Exception as error:
print(f"::error::NAT functional check failed: {error}", file=sys.stderr)
Expand Down
7 changes: 7 additions & 0 deletions .github/workflows/blueprint-qa.yml
Original file line number Diff line number Diff line change
Expand Up @@ -82,6 +82,12 @@ jobs:
docker compose -f docker-compose.infra.yml -f docker-compose.yml exec -T merchant \
python - < .github/scripts/check_nat_agents.py

# The functional probe performs live public-inference calls. Give the
# shared endpoint a brief cooldown before the independent browser suite
# starts its own Apps SDK searches and recommendations.
- name: Cool down hosted inference before UI tests
run: sleep 15

- name: Run UI automation tests
env:
TEST_DOCKER_PULL_KEY: ${{ secrets.TEST_DOCKER_PULL_KEY }}
Expand All @@ -93,6 +99,7 @@ jobs:
docker run --rm --net=host -v "$(pwd):/workspace" "${QA_TEST_IMAGE}" \
pytest -m retail_agentic_commerce \
--retail-agentic-commerce-url="${UI_URL}" \
--reruns=2 --reruns-delay=15 \
--disable-warnings -v --timeout=240 \
--html=/workspace/retail-agentic-commerce-ui-test.html --self-contained-html

Expand Down
10 changes: 5 additions & 5 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -84,8 +84,8 @@ flowchart TB
end

subgraph NIMs["NVIDIA NIMs"]
LLM["🧠 Nemotron Nano LLM<br/>(Port 8010)"]
EMBED["📐 NV-EmbedQA-E5<br/>(Port 8011)"]
LLM["🧠 Nemotron 3.5 Lightning LLM<br/>(Port 8010)"]
EMBED["📐 Nemotron 3 Embed 1B<br/>(Port 8011)"]
end

subgraph Data["Data Stores"]
Expand Down Expand Up @@ -174,8 +174,8 @@ Local NIM deployment requires NVIDIA GPUs to host the inference models. The foll

| Model | Purpose | Minimum GPU | Recommended GPU |
|-------|---------|-------------|-----------------|
| [Nemotron-Nano-30B-A3B](https://build.nvidia.com/nvidia/nemotron-3-nano-30b-a3b) | LLM — prompt planning, recommendations, search, promotions | 1× A100 (80 GB) | 1× H100 (80 GB) |
| [NV-EmbedQA-E5-v5](https://build.nvidia.com/nvidia/nv-embedqa-e5-v5) | Embedding — semantic search and product retrieval | 1× A100 (80 GB) | 1× H100 (80 GB) |
| [Nemotron-3.5-Lightning-30B-A3B](https://build.nvidia.com/nvidia/nemotron-3.5-lightning-30b-a3b) | LLM — prompt planning, recommendations, search, promotions | 1× A100 (80 GB) | 1× H100 (80 GB) |
| [Nemotron-3-Embed-1B](https://build.nvidia.com/nvidia/nemotron-3-embed-1b) | Embedding — semantic search and product retrieval | 1× A100 (80 GB) | 1× H100 (80 GB) |

**Total:** 2× A100 (80 GB) minimum, 2× H100 (80 GB) recommended for best performance.

Expand Down Expand Up @@ -221,6 +221,6 @@ docs/

## License

GOVERNING TERMS: The Blueprint scripts are governed by Apache License, Version 2.0, and enables use of separate open source and proprietary software governed by their respective licenses: [Nemotron-Nano-V3](https://catalog.ngc.nvidia.com/orgs/nim/teams/nvidia/containers/nemotron-3-nano?version=1.7.0), (ii) MIT license for [NV-EmbedQA-E5-v5](https://build.nvidia.com/nvidia/nv-embedqa-e5-v5). The sample data is governed by the [NVIDIA Data License for Retail Agentic Commerce](/LICENSE-assets.txt).
GOVERNING TERMS: The Blueprint scripts are governed by Apache License, Version 2.0, and enables use of separate open source and proprietary software governed by their respective licenses: [Nemotron-3.5-Lightning-30B-A3B](https://catalog.ngc.nvidia.com/orgs/nim/teams/nvidia/containers/nemotron-3.5-lightning-30b-a3b), (ii) MIT license for [Nemotron-3-Embed-1B](https://build.nvidia.com/nvidia/nemotron-3-embed-1b). The sample data is governed by the [NVIDIA Data License for Retail Agentic Commerce](/LICENSE-assets.txt).

This project will download and install additional third-party open source software projects. Review the license terms of these open source projects before use, found in [License-3rd-party.txt](/LICENSE-3rd-party.txt).
44 changes: 22 additions & 22 deletions deploy/1_Deploy_Agentic_Commerce.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -25,8 +25,8 @@
"**Deployment Mode: Local NIMs (On-Premises)**\n",
"\n",
"This deployment runs NVIDIA NIM microservices locally on your GPU infrastructure:\n",
"- **Nemotron Nano LLM** (`nvidia/nemotron-3-nano`) - For agent reasoning and decisions\n",
"- **NV-EmbedQA E5** (`nvidia/nv-embedqa-e5-v5`) - For product catalog embeddings and search\n",
"- **Nemotron 3.5 Lightning LLM** (`nvidia/nemotron-3.5-lightning`) - For agent reasoning and decisions\n",
"- **Nemotron 3 Embed 1B** (`nvidia/nemotron-3-embed-1b`) - For product catalog embeddings and search\n",
"\n",
"You will:\n",
"- Install the necessary prerequisites (including NVIDIA Container Toolkit for GPU access)\n",
Expand Down Expand Up @@ -302,11 +302,11 @@
"\n",
"| Service | Container | Port | Model |\n",
"|---------|-----------|------|-------|\n",
"| LLM | `nemotron-nano` | 8000 | `nvidia/nemotron-3-nano` |\n",
"| Embeddings | `embedqa` | 8000 | `nvidia/nv-embedqa-e5-v5` |\n",
"| LLM | `nemotron-lightning` | 8000 | `nvidia/nemotron-3.5-lightning` |\n",
"| Embeddings | `embedqa` | 8000 | `nvidia/nemotron-3-embed-1b` |\n",
"\n",
"<div class=\"alert alert-block alert-info\">\n",
" <b>Note:</b> The local NIM uses a different model name (<code>nvidia/nemotron-3-nano</code>) compared to the public endpoint (<code>nvidia/nemotron-3-nano-30b-a3b</code>).\n",
" <b>Note:</b> The local NIM uses <code>nvidia/nemotron-3.5-lightning</code>, while the public endpoint uses <code>nvidia/nemotron-3.5-lightning-30b-a3b</code>.\n",
"</div>"
]
},
Expand Down Expand Up @@ -355,17 +355,17 @@
"# Replace public NVIDIA API endpoints with local NIM container URLs\n",
"# =============================================================================\n",
"\n",
"# LLM Configuration (Nemotron Nano - local NIM)\n",
"# LLM Configuration (Nemotron 3.5 Lightning - local NIM)\n",
"content = content.replace(\n",
" \"NIM_LLM_BASE_URL=https://integrate.api.nvidia.com/v1\",\n",
" \"NIM_LLM_BASE_URL=http://nemotron-nano:8000/v1\"\n",
" \"NIM_LLM_BASE_URL=http://nemotron-lightning:8000/v1\"\n",
")\n",
"content = content.replace(\n",
" \"NIM_LLM_MODEL_NAME=nvidia/nemotron-3-nano-30b-a3b\",\n",
" \"NIM_LLM_MODEL_NAME=nvidia/nemotron-3-nano\"\n",
" \"NIM_LLM_MODEL_NAME=nvidia/nemotron-3.5-lightning-30b-a3b\",\n",
" \"NIM_LLM_MODEL_NAME=nvidia/nemotron-3.5-lightning\"\n",
")\n",
"\n",
"# Embedding Configuration (NV-EmbedQA - local NIM)\n",
"# Embedding Configuration (Nemotron 3 Embed 1B - local NIM)\n",
"content = content.replace(\n",
" \"NIM_EMBED_BASE_URL=https://integrate.api.nvidia.com/v1\",\n",
" \"NIM_EMBED_BASE_URL=http://embedqa:8000/v1\"\n",
Expand All @@ -383,10 +383,10 @@
"print(\" Recommendation Agent: http://recommendation-agent:8004\")\n",
"print(\" Search Agent: http://search-agent:8005\")\n",
"print(\"\\nNIM Configuration:\")\n",
"print(\" LLM Endpoint: http://nemotron-nano:8000/v1\")\n",
"print(\" LLM Model: nvidia/nemotron-3-nano\")\n",
"print(\" LLM Endpoint: http://nemotron-lightning:8000/v1\")\n",
"print(\" LLM Model: nvidia/nemotron-3.5-lightning\")\n",
"print(\" Embed Endpoint: http://embedqa:8000/v1\")\n",
"print(\" Embed Model: nvidia/nv-embedqa-e5-v5\")"
"print(\" Embed Model: nvidia/nemotron-3-embed-1b\")"
]
},
{
Expand All @@ -398,9 +398,9 @@
" To switch to <b>public NVIDIA API endpoints</b> instead of local NIMs, update your <code>.env</code> file with:\n",
" <pre>\n",
"NIM_LLM_BASE_URL=https://integrate.api.nvidia.com/v1\n",
"NIM_LLM_MODEL_NAME=nvidia/nemotron-3-nano-30b-a3b\n",
"NIM_LLM_MODEL_NAME=nvidia/nemotron-3.5-lightning-30b-a3b\n",
"NIM_EMBED_BASE_URL=https://integrate.api.nvidia.com/v1\n",
"NIM_EMBED_MODEL_NAME=nvidia/nv-embedqa-e5-v5\n",
"NIM_EMBED_MODEL_NAME=nvidia/nemotron-3-embed-1b\n",
" </pre>\n",
" And skip starting <code>docker-compose-nim.yml</code> (Step 1 below). No GPU required for public endpoint mode.\n",
"</div>"
Expand All @@ -424,8 +424,8 @@
"<div class=\"alert alert-block alert-info\">\n",
" <b>NIM Configuration:</b> The application services are configured via the <code>.env</code> file to connect to local NIM containers:\n",
" <ul>\n",
" <li><code>NIM_LLM_BASE_URL=http://nemotron-nano:8000/v1</code></li>\n",
" <li><code>NIM_LLM_MODEL_NAME=nvidia/nemotron-3-nano</code></li>\n",
" <li><code>NIM_LLM_BASE_URL=http://nemotron-lightning:8000/v1</code></li>\n",
" <li><code>NIM_LLM_MODEL_NAME=nvidia/nemotron-3.5-lightning</code></li>\n",
" <li><code>NIM_EMBED_BASE_URL=http://embedqa:8000/v1</code></li>\n",
" </ul>\n",
"</div>"
Expand Down Expand Up @@ -551,7 +551,7 @@
"apps-sdk 2026-02-03 01:08:39 +0000 UTC Up 2 minutes (healthy)\n",
"merchant 2026-02-03 01:08:39 +0000 UTC Up 2 minutes (healthy)\n",
"psp 2026-02-03 01:08:39 +0000 UTC Up 2 minutes (healthy)\n",
"nemotron-nano 2026-02-03 01:07:13 +0000 UTC Up 3 minutes (health: starting)\n",
"nemotron-lightning 2026-02-03 01:07:13 +0000 UTC Up 3 minutes (health: starting)\n",
"embedqa 2026-02-03 01:07:13 +0000 UTC Up 3 minutes (healthy)\n",
"milvus-standalone 2026-02-03 01:04:18 +0000 UTC Up 6 minutes (healthy)\n",
"milvus-minio 2026-02-03 01:04:16 +0000 UTC Up 7 minutes (healthy)\n",
Expand All @@ -564,7 +564,7 @@
"</div>\n",
"\n",
"<div class=\"alert alert-block alert-success\">\n",
" <b>Local NIMs:</b> The <code>nemotron-nano</code> and <code>embedqa</code> containers are the local NVIDIA NIMs running on your GPUs. These provide LLM and embedding inference to the NAT agents without requiring external API calls.\n",
" <b>Local NIMs:</b> The <code>nemotron-lightning</code> and <code>embedqa</code> containers are the local NVIDIA NIMs running on your GPUs. These provide LLM and embedding inference to the NAT agents without requiring external API calls.\n",
"</div>"
]
},
Expand Down Expand Up @@ -682,11 +682,11 @@
"\n",
"| Service | Container | External Port | Internal Port | Model |\n",
"|---------|-----------|---------------|---------------|-------|\n",
"| Nemotron Nano LLM | `nemotron-nano` | 8010 | 8000 | `nvidia/nemotron-3-nano` |\n",
"| Embedding Model | `embedqa` | 8011 | 8000 | `nvidia/nv-embedqa-e5-v5` |\n",
"| Nemotron 3.5 Lightning LLM | `nemotron-lightning` | 8010 | 8000 | `nvidia/nemotron-3.5-lightning` |\n",
"| Embedding Model | `embedqa` | 8011 | 8000 | `nvidia/nemotron-3-embed-1b` |\n",
"\n",
"<div class=\"alert alert-block alert-info\">\n",
" <b>Note:</b> NAT agents connect to NIMs via internal Docker network using container names (e.g., <code>http://nemotron-nano:8000/v1</code>). External ports (8010, 8011) are for health checks and debugging.\n",
" <b>Note:</b> NAT agents connect to NIMs via internal Docker network using container names (e.g., <code>http://nemotron-lightning:8000/v1</code>). External ports (8010, 8011) are for health checks and debugging.\n",
"</div>\n",
"\n",
"### Application Services\n",
Expand Down
13 changes: 9 additions & 4 deletions deploy/docker-deployment.md
Original file line number Diff line number Diff line change
Expand Up @@ -18,12 +18,17 @@ By default the stack calls NVIDIA public NIMs hosted on `build.nvidia.com`.
export NVIDIA_API_KEY=<YOUR_KEY>
```

**Optional** — override the model or point to self-hosted NIMs:
**Optional** — choose a hosted endpoint or point the stack to the local NIM containers:

```bash
export NIM_LLM_MODEL_NAME=nvidia/nemotron-3-nano-30b-a3b
export NIM_LLM_BASE_URL=http://HOST:POST/v1
export NIM_EMBED_BASE_URL=http://HOST:PORT/v1
# NVIDIA hosted endpoint
export NIM_LLM_MODEL_NAME=nvidia/nemotron-3.5-lightning-30b-a3b

# Local NIM containers from docker-compose-nim.yml
export NIM_LLM_BASE_URL=http://nemotron-lightning:8000/v1
export NIM_LLM_MODEL_NAME=nvidia/nemotron-3.5-lightning
export NIM_EMBED_BASE_URL=http://embedqa:8000/v1
export NIM_EMBED_MODEL_NAME=nvidia/nemotron-3-embed-1b
```

## 2. Create Shared Docker Network (one-time)
Expand Down
Loading
Loading