diff --git a/CHANGELOG.md b/CHANGELOG.md index 7b35b49e..bbc17709 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -38,6 +38,10 @@ All notable changes to SkillEvaluator are documented in this file. - Tier 3 log converters now rebuild ATIF trajectories from OpenCode JSON streams (`opencode.txt`) and structured Codex tee logs (`codex.txt`) when `trajectory.json` is missing or empty. +- `--env-mode runta` and `--env-mode mosaic` run Tier 3 trials on Harbor's Runta + and Mosaic backends. Runta's `mode` and Mosaic's `volume`, `persist`, + `enable_ssh`, `build_args`, and `build_target` kwargs are reserved because they + bypass the task image or let state and access outlive one isolated trial. ### Changed @@ -55,10 +59,10 @@ All notable changes to SkillEvaluator are documented in this file. Native tasks' own test scripts must write numeric `reward.json` values, which Harbor 0.24 enforces; a failed job now names the trial or step exception that Harbor recorded. -- Exposed 23 Harbor 0.24 backends alongside local mode. `cua-cloud`, - `opensandbox`, `hf-sandbox`, `podman`, `kata`, `runta`, `prime`, `mosaic`, - and `smol` remain disabled until generated tasks can be projected through a - trusted image or backend-native provisioning path. Non-secret backend +- Exposed 25 Harbor 0.24 backends alongside local mode. `cua-cloud`, + `opensandbox`, `hf-sandbox`, `podman`, `kata`, `prime`, and `smol` remain + disabled until generated tasks can be projected through a trusted image or + backend-native provisioning path. Non-secret backend constructor options can be supplied with repeatable, operator-only `--environment-kwarg` / `--ek` flags; skill-owned configuration, credentials, and sandbox-policy overrides remain outside that surface. diff --git a/docs/agents-and-sandboxes.mdx b/docs/agents-and-sandboxes.mdx index 6e8aef3f..b821f1b5 100644 --- a/docs/agents-and-sandboxes.mdx +++ b/docs/agents-and-sandboxes.mdx @@ -127,7 +127,7 @@ Anthropic evaluator with `opencode` also does not support local mode. ## Where trials run -SkillEvaluator exposes 24 environment modes: 23 Harbor-native backends, +SkillEvaluator exposes 26 environment modes: 25 Harbor-native backends, including Docker, plus SkillEvaluator's own `local` host-execution mode. Docker is the default. The `tier3` extra installs base `harbor==0.24.0`; install the environment extra shown below when a managed backend needs one. @@ -157,6 +157,8 @@ the environment extra shown below when a managed backend needs one. | `skypilot` | Harbor-native | `harbor[skypilot]==0.24.0` | | `hyperbrowser` | Harbor-native | `harbor[hyperbrowser]==0.24.0` | | `vercel` | Harbor-native | `harbor[vercel]==0.24.0` | +| `runta` | Harbor-native | `harbor[runta]==0.24.0` | +| `mosaic` | Harbor-native | `harbor[mosaic]==0.24.0` | | `local` | SkillEvaluator | Your machine, under an OS sandbox policy — see [Local mode](#local-mode). | @@ -173,12 +175,16 @@ authentication (`auth=wandb`), using `WANDB_API_KEY` or the netrc file that is rejected for both `cwsandbox` and `wandb`. Harbor 0.24 also defines `cua-cloud`, `opensandbox`, `hf-sandbox`, `podman`, -`kata`, `runta`, `prime`, `mosaic`, and `smol`, but those backends are not -exposed until SkillEvaluator can project the complete task bundle — skill, -verifier, dependencies, and inputs — through a trusted image or -backend-native provisioning path, with credentials and constructor options -reviewed per backend. Selecting one of those names is rejected before a run -starts. +`kata`, `prime`, and `smol`, but those backends are not exposed until +SkillEvaluator can project the complete task bundle — skill, verifier, +dependencies, and inputs — through a trusted image or backend-native +provisioning path, with credentials and constructor options reviewed per +backend. Selecting one of those names is rejected before a run starts. Prime +cannot build a Dockerfile-only task, which is how SkillEvaluator stages most +tasks, and Smol runs only a published `[environment].docker_image`, which +SkillEvaluator rejects because it bypasses the evaluator-staged image. Exposing +either one needs Compose-only staging or a trusted image-publishing step, +validated live. ### Native backend constructor options @@ -254,6 +260,37 @@ Harbor 0.24 Modal preflight requires `~/.modal.toml` or `MODAL_TOKEN_ID` plus custom config path by itself; configure one of Harbor's supported credential forms before launching trials. +Runta authenticates with `RUNTA_TOKEN` or the configuration file named by +`RUNTA_CONFIG` (default `~/.config/runta/config.toml`), and Harbor's preflight +requires one of the two variables. `RUNTA_ENDPOINT` selects a non-default API +endpoint. A relative `RUNTA_CONFIG` path is resolved against the directory you +run SkillEvaluator from, and a configured file must exist. Runta's `mode` +kwarg is reserved: `mode=direct` runs commands in Runta's base runtime and +ignores the task image, while Harbor's automatic selection builds the task's +Dockerfile or Compose definition inside the runtime. + +Mosaic authenticates with `MOSAIC_API_TOKEN` or the token `mos login` saves in +`~/.config/mosaic-sandbox/config.json`; `MOSAIC_CONFIG` names a different file, +which must exist and is resolved against the directory you run SkillEvaluator +from when relative. +`MOSAIC_API_URL` and `MOSAIC_RETRIES` tune the API client, the SDK's older +`MAR_API_TOKEN`, `MAR_ENDPOINT`, and `MAR_CONFIG` names are forwarded too, and +`MOSAIC_REGISTRY_USERNAME` and `MOSAIC_REGISTRY_PASSWORD` cover private base +images. Mosaic's `volume`, `persist`, and `enable_ssh` kwargs are +reserved because they let state or access outlive one isolated trial, and +`build_args` and `build_target` are reserved because they can skip or alter +evaluator-staged Dockerfile layers. Mosaic's `secrets` kwarg is rejected because +it injects provider secrets into the agent's sandbox, and a Mosaic token stored +only in `E2B_API_KEY` is rejected because that variable is never forwarded to +Mosaic. Mosaic runs one VM per trial, so tasks with an +`environment/docker-compose.yaml` are rejected before the run starts. + +Harbor builds each Mosaic task image into a snapshot named from the task and its +build context, reuses an existing snapshot with that name in later runs, and +never deletes snapshots. Run evaluations in a Mosaic account that only trusted +operators can write to, and delete old `harbor-*` snapshots when you no longer +need the staged skills and inputs they contain. + Other eligible, backend-specific constructor options are forwarded after validation. A key that the selected backend's Harbor 0.24.0 constructor does not accept is rejected before the run starts; for example, Harbor 0.24 removed diff --git a/docs/cli-reference.mdx b/docs/cli-reference.mdx index 10de1f8e..8ac15012 100644 --- a/docs/cli-reference.mdx +++ b/docs/cli-reference.mdx @@ -386,7 +386,7 @@ Use `skillevaluator tier3 PATH` for the complete workflow with automatic missing | Flag | Default | Effect | | --- | --- | --- | | `-a, --agents TEXT` | provider-native | Comma-separated Harbor agents. Defaults: NVIDIA Build=`opencode`, OpenAI=`codex`, Anthropic=`claude-code`. Supported: `claude-code`, `codex`, `opencode`; the alias `claude` is accepted for `claude-code`. See [Agents & Sandboxes](agents-and-sandboxes.mdx). | -| `--env-mode` | `docker` | Where trials run. SkillEvaluator exposes 24 environment modes: 23 Harbor-native modes — `docker`, `daytona`, `e2b`, `modal`, `runloop`, `langsmith`, `ec2`, `gke`, `ack`, `openshift`, `novita`, `apple-container`, `singularity`, `islo`, `tensorlake`, `cwsandbox`, `wandb`, `use-computer`, `blaxel`, `beam`, `skypilot`, `hyperbrowser`, `vercel` — plus SkillEvaluator `local` mode. `wandb` runs Harbor's `cwsandbox` backend with W&B authentication, so it needs the `harbor[cwsandbox]==0.24.0` extra; Harbor 0.24 has no `wandb` extra. Harbor's `cua-cloud`, `opensandbox`, `hf-sandbox`, `podman`, `kata`, `runta`, `prime`, `mosaic`, and `smol` backends are not selectable. See the complete extras and prerequisites matrix in [Agents & Sandboxes](agents-and-sandboxes.mdx#where-trials-run). | +| `--env-mode` | `docker` | Where trials run. SkillEvaluator exposes 26 environment modes: 25 Harbor-native modes — `docker`, `daytona`, `e2b`, `modal`, `runloop`, `langsmith`, `ec2`, `gke`, `ack`, `openshift`, `novita`, `apple-container`, `singularity`, `islo`, `tensorlake`, `cwsandbox`, `wandb`, `use-computer`, `blaxel`, `beam`, `skypilot`, `hyperbrowser`, `vercel`, `runta`, `mosaic` — plus SkillEvaluator `local` mode. `wandb` runs Harbor's `cwsandbox` backend with W&B authentication, so it needs the `harbor[cwsandbox]==0.24.0` extra; Harbor 0.24 has no `wandb` extra. Harbor's `cua-cloud`, `opensandbox`, `hf-sandbox`, `podman`, `kata`, `prime`, and `smol` backends are not selectable. See the complete extras and prerequisites matrix in [Agents & Sandboxes](agents-and-sandboxes.mdx#where-trials-run). | | `--environment-kwarg`, `--ek` | none | Operator-only `KEY=VALUE` constructor option for Harbor-native non-Docker backends (repeatable). Values accept JSON-compatible types; skill-owned `evals/config.yml` cannot set them. Never pass secrets. Docker and local reject all kwargs. Harbor runtime-policy fields for overrides, mounts, networks, extra Compose, lifecycle, and pod security are reserved, as are `stream`, `enable_environment_dir_upload`, and `auth` for `cwsandbox`/`wandb`. Names the selected backend does not accept in Harbor 0.24.0 are rejected. See [Native backend constructor options](agents-and-sandboxes.mdx#native-backend-constructor-options). | | `--autopilot` | off | Create one eval case when no dataset/task source exists, then evaluate. The case is LLM-generated with the configured provider, with a deterministic keyless template fallback; an existing source is never overwritten. | | `--skip-baseline` | off | Skip the without-skill baseline (no lift analysis, faster). | @@ -519,7 +519,7 @@ skillevaluator doctor --env-mode docker | Flag | Default | Effect | | --- | --- | --- | | `-a, --agents TEXT` | provider-native | Comma-separated agents to check. Defaults: NVIDIA Build=`opencode`, OpenAI=`codex`, Anthropic=`claude-code`. | -| `--env-mode` | `docker` | Backend to check (same 24 modes as [tier3 evaluate](#tier3-evaluate)). | +| `--env-mode` | `docker` | Backend to check (same 26 modes as [tier3 evaluate](#tier3-evaluate)). | | `--environment-kwarg`, `--ek` | none | `KEY=VALUE` non-secret constructor option for a Harbor-native non-Docker backend (repeatable). Pass the same values intended for evaluation. Docker and local reject all kwargs; policy-owned fields are rejected for every eligible backend. | | `--agent-model TEXT` | unset | Per-agent model override, `AGENT=MODEL` (repeatable) — check readiness with the model each agent will actually run. | | `--verify-models` | off | Live per-agent catalog probe. Prints pass for verified access, warn for an inconclusive or non-authoritative result (success or failure), and fail only for a definitive credential, model, or configuration rejection. | @@ -537,7 +537,7 @@ skillevaluator health-check | Flag | Default | Effect | | --- | --- | --- | | `-a, --agents TEXT` | provider-native | Comma-separated agents to check. Defaults: NVIDIA Build=`opencode`, OpenAI=`codex`, Anthropic=`claude-code`. | -| `--env-mode` | `docker` | Backend to check (same 24 modes as [tier3 evaluate](#tier3-evaluate)). | +| `--env-mode` | `docker` | Backend to check (same 26 modes as [tier3 evaluate](#tier3-evaluate)). | | `--environment-kwarg`, `--ek` | none | `KEY=VALUE` non-secret constructor option for a Harbor-native non-Docker backend (repeatable). Pass the same values intended for evaluation. Docker and local reject all kwargs; policy-owned fields are rejected for every eligible backend. | ## models diff --git a/docs/tier3-live-evaluation.mdx b/docs/tier3-live-evaluation.mdx index 0d6390b9..f1c6622f 100644 --- a/docs/tier3-live-evaluation.mdx +++ b/docs/tier3-live-evaluation.mdx @@ -390,7 +390,7 @@ explicit models — see [Agents & Sandboxes](agents-and-sandboxes.mdx). ## Where agents run -`--env-mode` selects one of 24 environment modes: 23 Harbor-native backends, +`--env-mode` selects one of 26 environment modes: 25 Harbor-native backends, including the default Docker backend, plus SkillEvaluator's **local mode**, which runs the agent CLI directly on the host under an OS-sandbox policy. Managed backends such as `daytona`, `e2b`, and `modal` require their mapped @@ -402,8 +402,8 @@ validate any managed backend's extra, credentials, infrastructure, and constructor settings with `doctor` and its Harbor preflight before relying on it. -Harbor's `cua-cloud`, `opensandbox`, `hf-sandbox`, `podman`, `kata`, `runta`, -`prime`, `mosaic`, and `smol` backends are not selectable because they cannot +Harbor's `cua-cloud`, `opensandbox`, `hf-sandbox`, `podman`, `kata`, `prime`, +and `smol` backends are not selectable because they cannot yet receive SkillEvaluator's complete generated task bundle through a trusted image or backend-native provisioning path. diff --git a/src/skillevaluator/tier3/harbor/runner.py b/src/skillevaluator/tier3/harbor/runner.py index c46d0f16..4c186822 100644 --- a/src/skillevaluator/tier3/harbor/runner.py +++ b/src/skillevaluator/tier3/harbor/runner.py @@ -23,7 +23,7 @@ import threading import time import tomllib -from collections.abc import Iterator, Mapping +from collections.abc import Iterable, Iterator, Mapping from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, as_completed, wait from contextlib import ExitStack, contextmanager, suppress from dataclasses import dataclass, field @@ -889,6 +889,21 @@ def _reserve_run_dir(results_root: Path, timestamp: str) -> Path: ), "hyperbrowser": _DOCKER_HOST_ENV_VARS | frozenset({"HYPERBROWSER_API_KEY", "HYPERBROWSER_BASE_URL"}), "vercel": frozenset({"VERCEL_OIDC_TOKEN", "VERCEL_PROJECT_ID", "VERCEL_TEAM_ID", "VERCEL_TOKEN"}), + "runta": frozenset({"RUNTA_CONFIG", "RUNTA_ENDPOINT", "RUNTA_TOKEN"}), + # The Mosaic SDK still reads the MAR_* names its MOSAIC_* names replaced. + "mosaic": frozenset( + { + "MAR_API_TOKEN", + "MAR_CONFIG", + "MAR_ENDPOINT", + "MOSAIC_API_TOKEN", + "MOSAIC_API_URL", + "MOSAIC_CONFIG", + "MOSAIC_REGISTRY_PASSWORD", + "MOSAIC_REGISTRY_USERNAME", + "MOSAIC_RETRIES", + } + ), } _BEDROCK_HOST_ENV_VARS = _AWS_HOST_ENV_VARS | { "AWS_BEARER_TOKEN_BEDROCK", @@ -961,9 +976,12 @@ def _harbor_bin() -> str: "DOCKER_CONFIG", "GOOGLE_APPLICATION_CREDENTIALS", "LANGSMITH_CONFIG_FILE", + "MAR_CONFIG", "MODAL_CONFIG_PATH", + "MOSAIC_CONFIG", "NETRC", "REQUESTS_CA_BUNDLE", + "RUNTA_CONFIG", "SINGULARITY_AUTHFILE", "SINGULARITY_CONFIGDIR", "SKYPILOT_GLOBAL_CONFIG", @@ -1124,7 +1142,14 @@ def _nvidia_build_key_handoff( "ec2": frozenset({"iam_instance_profile", "strict_host_key_checking"}), "gke": frozenset({"memory_limit_multiplier"}), "modal": frozenset({"volumes"}), + # Persistent volumes, persisted sandboxes, SSH access, and alternate build + # inputs would let state or access outlive one isolated trial or skip + # evaluator-staged Dockerfile layers. + "mosaic": frozenset({"build_args", "build_target", "enable_ssh", "persist", "volume"}), "openshift": frozenset({"service_account_name"}), + # Direct mode runs commands in Runta's base runtime and ignores the task + # image; Harbor's automatic selection honors the task's Dockerfile or Compose. + "runta": frozenset({"mode"}), "singularity": frozenset({"singularity_no_mount"}), "use-computer": frozenset({"resources"}), "vercel": frozenset({"ports"}), @@ -1566,6 +1591,62 @@ def invalid_strings(*names: str) -> list[str]: return [] +# Backend SDK configuration files that operators may name in the host environment. +_BACKEND_CONFIG_FILE_ENV_VARS: Mapping[str, tuple[str, ...]] = MappingProxyType( + {"mosaic": ("MOSAIC_CONFIG", "MAR_CONFIG"), "runta": ("RUNTA_CONFIG",)} +) + + +def _backend_config_prerequisite_errors(env_mode: str) -> list[str]: + """Check Runta and Mosaic credential inputs against what Harbor's child will see.""" + errors: list[str] = [] + for name in _BACKEND_CONFIG_FILE_ENV_VARS.get(env_mode, ()): + raw = os.environ.get(name) + if raw is None: + continue + try: + is_file = bool(raw.strip()) and Path(raw).expanduser().is_file() + except (OSError, RuntimeError): + is_file = False + if not is_file: + errors.append(f"Harbor environment '{env_mode}' requires {name} to name an existing regular file.") + if env_mode == "mosaic" and not errors: + try: + from mosaic_sandbox.credentials import resolve_token + except ImportError: + return errors # Harbor's preflight reports the missing extra. + _token, source = resolve_token() + # The SDK accepts a Mosaic token in E2B_API_KEY for E2B migrations, but + # that variable belongs to the e2b backend and never reaches Mosaic. + if source == "e2b_environment": + errors.append( + "Harbor environment 'mosaic' does not forward E2B_API_KEY; set MOSAIC_API_TOKEN or run `mos login`." + ) + return errors + + +def _staged_task_environment_error(env_mode: str, task_dirs: Iterable[Path]) -> str | None: + """Reject staged task environments the selected backend cannot run.""" + if env_mode != "mosaic": + return None + for task_dir in task_dirs: + if (task_dir / "environment" / "docker-compose.yaml").exists(): + return ( + f"Harbor environment 'mosaic' runs one VM per trial and cannot run Docker Compose task " + f"'{task_dir.name}'; use a Dockerfile-only environment or another backend." + ) + return None + + +def _harbor_missing_dependency_summary(exc: ImportError) -> str: + """Keep Harbor's diagnosis, without its unpinned install commands or cloud-bundle hint.""" + lines = str(exc).strip().splitlines() + summary = lines[0] if lines else type(exc).__name__ + for suffix in (" Install it with:", " Install them with:"): + summary = summary.removesuffix(suffix) + return summary.strip().rstrip(".") + + def _cwsandbox_prerequisite_errors(env_mode: str) -> list[str]: """Check what Harbor's cwsandbox backend needs; Harbor 0.24 no longer checks it.""" if harbor_environment_type(env_mode) != "cwsandbox": @@ -1838,6 +1919,8 @@ def _check_prerequisites( if cwsandbox_errors := _cwsandbox_prerequisite_errors(env_mode): return cwsandbox_errors + if backend_config_errors := _backend_config_prerequisite_errors(env_mode): + return backend_config_errors EnvironmentFactory.run_preflight(EnvironmentType(harbor_environment_type(env_mode))) if env_mode == "ack": ack_subprocess_env = ( @@ -1854,7 +1937,7 @@ def _check_prerequisites( ) except ImportError as exc: detail = redact_progress_detail( - exc, + _harbor_missing_dependency_summary(exc), secret_values=secret_values_from_environment(os.environ), ) return [ @@ -4167,6 +4250,8 @@ def _persist_pre_execution_failure(errors: list[str]) -> dict[str, Any]: evaluator_skill_path=evaluator_skill_path, arm_suffix=with_arm_suffix, ) + if staged_environment_error := _staged_task_environment_error(env_mode, task_paths): + raise ValueError(staged_environment_error) task_selectors = validate_case_ids(task.name for task in task_paths) logical_case_ids = validate_case_ids(_native_entry_id(task) for task in task_paths) case_id_by_task_selector = dict(zip(task_selectors, logical_case_ids, strict=True)) diff --git a/src/skillevaluator/tier3_environments.py b/src/skillevaluator/tier3_environments.py index 60f1a300..4ddcb3f5 100644 --- a/src/skillevaluator/tier3_environments.py +++ b/src/skillevaluator/tier3_environments.py @@ -37,6 +37,8 @@ "skypilot", "hyperbrowser", "vercel", + "runta", + "mosaic", # Not a Harbor-native backend: SkillEvaluator's host execution mode, run # under an OS sandbox (bubblewrap on Linux, Seatbelt on macOS). Dispatched # by passing its custom import path through Harbor's unified --env flag. @@ -85,14 +87,16 @@ def harbor_environment_type(env_mode: str) -> str: "skypilot": "skypilot", "hyperbrowser": "hyperbrowser", "vercel": "vercel", + "runta": "runta", + "mosaic": "mosaic", } # Constructor kwargs consumed by the pinned Harbor release. Keep this static so # importing the base SkillEvaluator CLI never imports Harbor or optional provider # SDKs. The packaging parity test AST-reads the pinned Harbor sources and catches # additions, removals, and provider kwargs that are consumed through **kwargs. -# Registry-only entries (cua-cloud, opensandbox, hf-sandbox, podman, kata, runta, -# prime, mosaic, smol) keep that parity exact; those modes are not exposed. +# Registry-only entries (cua-cloud, opensandbox, hf-sandbox, podman, kata, prime, +# smol) keep that parity exact; those modes are not exposed. _HARBOR_BASE_ENVIRONMENT_KWARGS = frozenset( { "cpu_enforcement_policy", diff --git a/tests/golden/cli_surface.json b/tests/golden/cli_surface.json index d51f092a..cc2eb51f 100644 --- a/tests/golden/cli_surface.json +++ b/tests/golden/cli_surface.json @@ -274,6 +274,8 @@ "skypilot", "hyperbrowser", "vercel", + "runta", + "mosaic", "local" ], "default": "docker", @@ -359,6 +361,8 @@ "skypilot", "hyperbrowser", "vercel", + "runta", + "mosaic", "local" ], "default": "docker", @@ -663,6 +667,8 @@ "skypilot", "hyperbrowser", "vercel", + "runta", + "mosaic", "local" ], "default": "docker", @@ -2127,6 +2133,8 @@ "skypilot", "hyperbrowser", "vercel", + "runta", + "mosaic", "local" ], "default": "docker", @@ -2212,6 +2220,8 @@ "skypilot", "hyperbrowser", "vercel", + "runta", + "mosaic", "local" ], "default": "docker", @@ -2715,6 +2725,8 @@ "skypilot", "hyperbrowser", "vercel", + "runta", + "mosaic", "local" ], "default": "docker", @@ -3231,6 +3243,8 @@ "skypilot", "hyperbrowser", "vercel", + "runta", + "mosaic", "local" ], "default": "docker", diff --git a/tests/test_harbor_environment_contract.py b/tests/test_harbor_environment_contract.py index c85293cb..c3ef2078 100644 --- a/tests/test_harbor_environment_contract.py +++ b/tests/test_harbor_environment_contract.py @@ -7,6 +7,8 @@ import inspect import json +import sys +import types from pathlib import Path import pytest @@ -85,6 +87,198 @@ def test_backend_kwargs_track_the_pinned_harbor_release() -> None: assert f"dind_image={json.dumps('docker:dind')}" in _environment_kwargs(tensorlake) +@pytest.mark.parametrize( + ("env_mode", "name", "value"), + [ + ("runta", "mode", "direct"), + ("mosaic", "volume", "shared-cache"), + ("mosaic", "persist", True), + ("mosaic", "enable_ssh", True), + ("mosaic", "build_args", {"BASE_IMAGE": "python:3.12"}), + ("mosaic", "build_target", "builder"), + ], +) +def test_runta_and_mosaic_isolation_controls_are_reserved(env_mode: str, name: str, value: object) -> None: + with pytest.raises(ValueError, match=rf"reserved for Harbor runtime policy: {name}"): + build_harbor_run_command( + dataset_path="/tmp/dataset", + agent="opencode", + job_name="reserved", + env_mode=env_mode, + environment_kwargs={name: value}, + ) + + +@pytest.mark.parametrize(("env_mode", "name", "value"), [("runta", "token", "rt-123456"), ("mosaic", "secrets", ["s"])]) +def test_runta_and_mosaic_credentials_stay_in_the_host_environment(env_mode: str, name: str, value: object) -> None: + with pytest.raises(ValueError, match="secret-bearing"): + build_harbor_run_command( + dataset_path="/tmp/dataset", + agent="opencode", + job_name="credentials", + env_mode=env_mode, + environment_kwargs={name: value}, + ) + + +def test_runta_and_mosaic_forward_operational_kwargs() -> None: + runta = build_harbor_run_command( + dataset_path="/tmp/dataset", + agent="opencode", + job_name="runta", + env_mode="runta", + environment_kwargs={"endpoint": "https://runta.example", "startup_timeout_sec": 300}, + ) + assert runta[runta.index("--env") + 1] == "runta" + assert _environment_kwargs(runta) == [f"endpoint={json.dumps('https://runta.example')}", "startup_timeout_sec=300"] + + mosaic = build_harbor_run_command( + dataset_path="/tmp/dataset", + agent="opencode", + job_name="mosaic", + env_mode="mosaic", + environment_kwargs={"metadata": {"team": "evals"}, "replicas": 2, "ttl_seconds": 7200}, + ) + assert mosaic[mosaic.index("--env") + 1] == "mosaic" + assert _environment_kwargs(mosaic) == [ + f"metadata={json.dumps({'team': 'evals'}, separators=(',', ':'))}", + "replicas=2", + "ttl_seconds=7200", + ] + + +def test_runta_and_mosaic_config_paths_survive_the_evaluator_owned_launch_directory( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + monkeypatch.chdir(tmp_path) + + launch_env = runner._harbor_launch_environment( + { + "RUNTA_CONFIG": "runta/config.toml", + "MOSAIC_CONFIG": "mosaic/config.json", + "MAR_CONFIG": "~/.config/mosaic-sandbox/config.json", + "MOSAIC_API_URL": "https://sandbox.example", + } + ) + + assert launch_env["RUNTA_CONFIG"] == str(tmp_path / "runta" / "config.toml") + assert launch_env["MOSAIC_CONFIG"] == str(tmp_path / "mosaic" / "config.json") + assert launch_env["MAR_CONFIG"] == "~/.config/mosaic-sandbox/config.json" + assert launch_env["MOSAIC_API_URL"] == "https://sandbox.example" + + +@pytest.mark.parametrize( + ("env_mode", "name"), + [ + ("mosaic", "MAR_ENDPOINT"), + ("mosaic", "MAR_CONFIG"), + ("mosaic", "MAR_API_TOKEN"), + ("mosaic", "MOSAIC_CONFIG"), + ("mosaic", "MOSAIC_API_URL"), + ("runta", "RUNTA_ENDPOINT"), + ("runta", "RUNTA_CONFIG"), + ], +) +def test_skill_runtime_env_cannot_redirect_runta_or_mosaic_credentials(env_mode: str, name: str) -> None: + resolved, errors = runner._resolve_runtime_env({name: "https://attacker.example"}, env_mode=env_mode) + + assert resolved == {} + assert errors == [f"harbor.runtime_env.{name} controls the host process and is not allowed"] + + +def test_mosaic_allowlist_covers_the_sdk_credential_names() -> None: + credentials = pytest.importorskip("mosaic_sandbox.credentials") + + for name in ("MOSAIC_API_TOKEN", "MOSAIC_API_URL", "MOSAIC_CONFIG"): + assert name in runner._HARBOR_ENV_MODE_VARS["mosaic"] + assert credentials.LEGACY_ENV_NAMES[name] in runner._HARBOR_ENV_MODE_VARS["mosaic"] + + +def test_runta_and_mosaic_credentials_do_not_cross_backends(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr( + runner.os, + "environ", + {"PATH": "/usr/bin", "HOME": "/home/test", "RUNTA_TOKEN": "runta-value", "MOSAIC_API_TOKEN": "mosaic-value"}, + ) + + runta = runner._selected_host_environment(runner._HARBOR_ENV_MODE_VARS["runta"], runner.os.environ) + mosaic = runner._selected_host_environment(runner._HARBOR_ENV_MODE_VARS["mosaic"], runner.os.environ) + + assert runta == {"RUNTA_TOKEN": "runta-value"} + assert mosaic == {"MOSAIC_API_TOKEN": "mosaic-value"} + + +@pytest.mark.parametrize( + ("env_mode", "name"), [("runta", "RUNTA_CONFIG"), ("mosaic", "MOSAIC_CONFIG"), ("mosaic", "MAR_CONFIG")] +) +def test_runta_and_mosaic_config_files_must_exist( + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, + env_mode: str, + name: str, +) -> None: + monkeypatch.chdir(tmp_path) + monkeypatch.setitem(sys.modules, "mosaic_sandbox.credentials", None) + monkeypatch.setenv(name, "missing/config.toml") + + assert runner._backend_config_prerequisite_errors(env_mode) == [ + f"Harbor environment '{env_mode}' requires {name} to name an existing regular file." + ] + + (tmp_path / "missing").mkdir() + (tmp_path / "missing" / "config.toml").write_text("token = 'x'\n", encoding="utf-8") + assert runner._backend_config_prerequisite_errors(env_mode) == [] + + +@pytest.mark.parametrize(("source", "rejected"), [("e2b_environment", True), ("environment", False), ("config", False)]) +def test_mosaic_rejects_a_token_only_in_e2b_api_key( + monkeypatch: pytest.MonkeyPatch, source: str, rejected: bool +) -> None: + fake = types.ModuleType("mosaic_sandbox.credentials") + fake.resolve_token = lambda: ("msk_live_value", source) # type: ignore[attr-defined] + monkeypatch.setitem(sys.modules, "mosaic_sandbox.credentials", fake) + for name in ("MOSAIC_CONFIG", "MAR_CONFIG"): + monkeypatch.delenv(name, raising=False) + + errors = runner._backend_config_prerequisite_errors("mosaic") + + assert bool(errors) is rejected + if rejected: + assert "E2B_API_KEY" in errors[0] + + +def test_mosaic_rejects_compose_tasks_before_harbor_starts(tmp_path: Path) -> None: + single = tmp_path / "single" + (single / "environment").mkdir(parents=True) + compose = tmp_path / "compose" + (compose / "environment").mkdir(parents=True) + (compose / "environment" / "docker-compose.yaml").write_text("services: {}\n", encoding="utf-8") + + assert runner._staged_task_environment_error("mosaic", [single]) is None + assert "cannot run Docker Compose task 'compose'" in runner._staged_task_environment_error( + "mosaic", [single, compose] + ) + assert runner._staged_task_environment_error("runta", [compose]) is None + + +def test_missing_extra_message_keeps_harbor_diagnosis_without_unpinned_install_hints() -> None: + from harbor.utils.optional_import import MissingExtraError + + summary = runner._harbor_missing_dependency_summary(MissingExtraError(package="mosaic-sandbox", extra="mosaic")) + + assert summary == "The 'mosaic-sandbox' package is required but not installed" + assert runner._harbor_missing_dependency_summary(ImportError("No module named 'runta'")) == ( + "No module named 'runta'" + ) + + +@pytest.mark.parametrize("env_mode", ["prime", "smol"]) +def test_harbor_backends_without_task_projection_stay_unexposed(env_mode: str) -> None: + assert env_mode not in runner.HARBOR_ENV_MODES + assert "Unsupported Harbor environment" in runner._check_prerequisites(env_mode=env_mode, agents=["opencode"])[0] + + def test_cwsandbox_prerequisites_require_sdk_and_api_key(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.setattr(runner.importlib.util, "find_spec", lambda name: None if name == "cwsandbox" else object()) assert "harbor[cwsandbox]==0.24.0" in runner._cwsandbox_prerequisite_errors("cwsandbox")[0] diff --git a/tests/test_harbor_local_mode.py b/tests/test_harbor_local_mode.py index d6e94b9f..f17a8569 100644 --- a/tests/test_harbor_local_mode.py +++ b/tests/test_harbor_local_mode.py @@ -465,9 +465,7 @@ def test_registered_native_env_modes_are_supported_subset_of_pinned_harbor_relea "hf-sandbox", "podman", "kata", - "runta", "prime", - "mosaic", "smol", } diff --git a/tests/test_harbor_runner_environment.py b/tests/test_harbor_runner_environment.py index 3a3a4b5c..5bd63cec 100644 --- a/tests/test_harbor_runner_environment.py +++ b/tests/test_harbor_runner_environment.py @@ -1840,6 +1840,18 @@ def test_harbor_backend_environment_allowlist_covers_every_native_022_mode() -> "SSH_AUTH_SOCK", }, "vercel": {"VERCEL_OIDC_TOKEN", "VERCEL_PROJECT_ID", "VERCEL_TEAM_ID", "VERCEL_TOKEN"}, + "runta": {"RUNTA_CONFIG", "RUNTA_ENDPOINT", "RUNTA_TOKEN"}, + "mosaic": { + "MAR_API_TOKEN", + "MAR_CONFIG", + "MAR_ENDPOINT", + "MOSAIC_API_TOKEN", + "MOSAIC_API_URL", + "MOSAIC_CONFIG", + "MOSAIC_REGISTRY_PASSWORD", + "MOSAIC_REGISTRY_USERNAME", + "MOSAIC_RETRIES", + }, } assert set(expected) == HARBOR_NATIVE_ENV_MODES @@ -2048,6 +2060,12 @@ def test_skip_baseline_keeps_normalized_skill_owned_setup_enabled( ("hyperbrowser", "DOCKER_HOST"), ("wandb", "NETRC"), ("vercel", "VERCEL_OIDC_TOKEN"), + ("runta", "RUNTA_TOKEN"), + ("runta", "RUNTA_ENDPOINT"), + ("mosaic", "MOSAIC_API_TOKEN"), + ("mosaic", "MOSAIC_API_URL"), + ("mosaic", "MOSAIC_CONFIG"), + ("mosaic", "MAR_API_TOKEN"), ], ) def test_new_harbor_022_backend_environment_is_selected_without_cross_backend_leakage( diff --git a/tests/test_oss_packaging.py b/tests/test_oss_packaging.py index ef0f214e..f4dd27f4 100644 --- a/tests/test_oss_packaging.py +++ b/tests/test_oss_packaging.py @@ -181,8 +181,8 @@ def test_harbor_environment_extra_mapping_matches_installed_metadata() -> None: provided_extras = set(metadata("harbor").get_all("Provides-Extra") or ()) system_or_base_backends = {"docker", "openshift", "apple-container", "singularity"} - assert len(HARBOR_ENVIRONMENTS) == 24 - assert len(HARBOR_NATIVE_ENV_MODES) == 23 + assert len(HARBOR_ENVIRONMENTS) == 26 + assert len(HARBOR_NATIVE_ENV_MODES) == 25 assert frozenset(HARBOR_ENVIRONMENTS) - {"local"} == HARBOR_NATIVE_ENV_MODES assert set(HARBOR_ENVIRONMENT_EXTRAS) == HARBOR_NATIVE_ENV_MODES assert {mode for mode, extra in HARBOR_ENVIRONMENT_EXTRAS.items() if extra is None} == system_or_base_backends @@ -316,8 +316,8 @@ def test_public_docs_match_harbor_environment_and_kwarg_contract() -> None: public_docs = f"{agents}\n{cli_reference}\n{configuration}\n{tier3}\n{eval_config}" normalized_docs = " ".join(public_docs.split()) - assert "24 environment modes" in public_docs - assert "23 Harbor-native backends" in public_docs + assert "26 environment modes" in public_docs + assert "25 Harbor-native backends" in public_docs assert "All 16 values" not in public_docs assert "same 16 values" not in public_docs assert "14 additional Harbor-native backends" not in public_docs