Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 13 additions & 2 deletions .github/workflows/ci.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -57,7 +57,7 @@ jobs:
run: uv sync --all-extras --no-extra leapspace

- name: Lint
run: uv run ruff check src/leapflow/ tests/ tools/
run: uv run ruff check src/leapflow/ src/benchmarks/ tests/benchmarks/ tests/ tools/

- name: Verify derived fixtures match the cassette store
run: uv run python tools/sync_fixtures.py --check
Expand All @@ -68,6 +68,17 @@ jobs:
- name: Mock layer (full)
run: uv run pytest tests/ -q -m "not e2e" --tb=short -n auto

- name: Benchmark contract suite
run: |
source .venv/bin/activate
PYTHONPATH=src python -m pytest tests/benchmarks -q

- name: Native benchmark smoke profiles
run: |
source .venv/bin/activate
PYTHONPATH=src python -m benchmarks run --profile tier0 --json
PYTHONPATH=src python -m benchmarks run --profile tier1 --json

# ── L2: main lane ──────────────────────────────────────────────────────
# Everything, unscoped, across the supported matrix.
# Coverage XML + artifact are generated only on the canonical environment
Expand Down Expand Up @@ -100,7 +111,7 @@ jobs:
run: uv sync --all-extras --no-extra leapspace

- name: Lint
run: uv run ruff check src/leapflow/ tests/ tools/
run: uv run ruff check src/leapflow/ src/benchmarks/ tests/benchmarks/ tests/ tools/

- name: Verify derived fixtures match the stored exchanges
run: uv run python tools/sync_fixtures.py --check
Expand Down
12 changes: 11 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -83,9 +83,15 @@ robot = [
robot-ros2 = [
"rclpy>=3.0; python_version>='3.11'",
]
benchmark = [
"gymnasium>=0.29",
"numpy>=1.24",
"pyarrow>=14.0",
]

[project.scripts]
leap = "leapflow.__main__:main"
leap-bench = "benchmarks.cli:main"

[build-system]
requires = ["setuptools>=68.0"]
Expand All @@ -96,13 +102,17 @@ version = {attr = "leapflow.version.__version__"}

[tool.setuptools.packages.find]
where = ["src"]
include = ["leapflow*", "leapspace*"]
include = ["leapflow*", "leapspace*", "benchmarks*"]

[tool.setuptools.package-data]
"leapflow.gateway.action_packs" = ["*.yaml"]
"leapflow.dashboard.templates" = ["*.yaml"]
"leapflow.dashboard.static" = ["*"]
"leapflow.plugins.dsh" = ["*.js"]
"benchmarks.manifests" = ["*.yaml"]
"benchmarks.manifests.native" = ["*.yaml"]
"benchmarks.manifests.external" = ["*.yaml"]
"benchmarks.manifests.hardware" = ["*.yaml"]
# LeapSpace task directories are data, not importable packages (note the
# non-identifier `task-001`). Auto-discovery already ships each task's
# action.py; this keeps its config.yaml beside it so an installed example task
Expand Down
66 changes: 66 additions & 0 deletions src/benchmarks/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
# Copyright (c) Alibaba, Inc. and its affiliates.
"""Standalone, import-safe benchmark harness for LeapFlow.

The package depends only on the Python standard library and PyYAML. Adapter
discovery is lazy; importing ``benchmarks`` never imports an external
benchmark SDK, starts a process, downloads data, or modifies the environment.
"""

from benchmarks.doctor import DoctorCheck, DoctorReport, run_doctor
from benchmarks.evidence import EvidenceStore
from benchmarks.gates import DEFAULT_THRESHOLDS, evaluate_gate
from benchmarks.manifest import load_manifest, load_manifests
from benchmarks.models import (
AvailabilityResult,
BenchmarkManifest,
BenchmarkResult,
EnvironmentFingerprint,
EvidenceRef,
GateResult,
GateStatus,
ManifestError,
MetricValue,
RunConfig,
Scenario,
TrialResult,
TrialStatus,
)
from benchmarks.profiles import BenchmarkProfile, get_profile, list_profiles
from benchmarks.protocol import BenchmarkAdapter, BenchmarkSuite, ScenarioProvider
from benchmarks.registry import AdapterConflict, AdapterRegistry
from benchmarks.runner import BenchmarkRunner, run_benchmark, run_benchmark_sync

__all__ = [
"AdapterConflict",
"AdapterRegistry",
"AvailabilityResult",
"BenchmarkAdapter",
"BenchmarkManifest",
"BenchmarkProfile",
"BenchmarkResult",
"BenchmarkRunner",
"BenchmarkSuite",
"DEFAULT_THRESHOLDS",
"DoctorCheck",
"DoctorReport",
"EnvironmentFingerprint",
"EvidenceRef",
"EvidenceStore",
"GateResult",
"GateStatus",
"ManifestError",
"MetricValue",
"RunConfig",
"Scenario",
"ScenarioProvider",
"TrialResult",
"TrialStatus",
"evaluate_gate",
"get_profile",
"list_profiles",
"load_manifest",
"load_manifests",
"run_benchmark",
"run_benchmark_sync",
"run_doctor",
]
7 changes: 7 additions & 0 deletions src/benchmarks/__main__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
# Copyright (c) Alibaba, Inc. and its affiliates.
"""Module entry point for ``python -m benchmarks``."""

from benchmarks.cli import main

if __name__ == "__main__":
raise SystemExit(main())
59 changes: 59 additions & 0 deletions src/benchmarks/adapters/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,59 @@
# Copyright (c) Alibaba, Inc. and its affiliates.
"""Lazy adapter factory for external benchmark adapters.

Import-safe: importing this module never imports an external benchmark
SDK. Adapters are instantiated on demand via ``builtin_adapters()``.
"""

from __future__ import annotations

import logging
from typing import Sequence

from benchmarks.protocol import BenchmarkAdapter

logger = logging.getLogger(__name__)


def builtin_adapters() -> Sequence[BenchmarkAdapter]:
"""Instantiate all built-in external benchmark adapters.

Each adapter is imported inside the function body so that a broken
adapter module cannot prevent others from loading.
"""
adapters: list[BenchmarkAdapter] = []

_ADAPTER_FACTORIES: tuple[tuple[str, str, str], ...] = (
("benchmarks.adapters.embodyguard", "EmBodyGuardAdapter", "embodyguard"),
("benchmarks.adapters.asimov", "AsimovAdapter", "asimov"),
("benchmarks.adapters.is_bench", "ISBenchAdapter", "is_bench"),
("benchmarks.adapters.kinder", "KinderAdapter", "kinder"),
("benchmarks.adapters.calvin", "CalvinAdapter", "calvin"),
("benchmarks.adapters.vlabench", "VLABenchAdapter", "vlabench"),
("benchmarks.adapters.robojailbench", "RoboJailBenchAdapter", "robojailbench"),
("benchmarks.adapters.attackvla", "AttackVLAAdapter", "attackvla"),
("benchmarks.adapters.safety_gymnasium", "SafetyGymnasiumAdapter", "safety_gymnasium"),
("benchmarks.adapters.maniskill", "ManiSkillAdapter", "maniskill"),
("benchmarks.adapters.isaac_lab", "IsaacLabAdapter", "isaac_lab"),
("benchmarks.adapters.external_command", "ExternalCommandAdapter", "external_command"),
("benchmarks.adapters.live_llm", "LiveLLMAdapter", "live_llm"),
("benchmarks.adapters.hardware_preflight", "HardwarePreflightAdapter", "hardware_preflight"),
)

for module_path, class_name, adapter_id in _ADAPTER_FACTORIES:
try:
import importlib
mod = importlib.import_module(module_path)
cls = getattr(mod, class_name)
adapter = cls()
adapters.append(adapter)
except Exception:
logger.warning(
"failed to load built-in adapter %r from %s",
adapter_id, module_path, exc_info=True,
)

return adapters


__all__ = ["builtin_adapters"]
92 changes: 92 additions & 0 deletions src/benchmarks/adapters/asimov.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
# Copyright (c) Alibaba, Inc. and its affiliates.
"""ASIMOV adapter — official semantic physical-safety benchmark.

ASIMOV evaluates alignment accuracy on its held-out Multimodal, Injury, and
Dilemmas safety datasets. It is a classification benchmark, not a synthetic
Three-Laws task generator.

Required: an operator-configured command from the verified benchmark homepage.
"""

from __future__ import annotations

from dataclasses import dataclass
from typing import Any, Mapping, Sequence

from benchmarks.adapters.base import (
command_availability,
load_adapter_manifest,
run_configured_command,
trial_from_subprocess,
unavailable_result,
)
from benchmarks.models import AvailabilityResult, Scenario, TrialResult

_ADAPTER_ID = "asimov"
_ADAPTER_VERSION = "0.1.0"
_COMMAND_ENV = "LEAPFLOW_BENCHMARK_ASIMOV_COMMAND"
_HOMEPAGE = "https://asimov-benchmark.github.io/"


@dataclass(frozen=True)
class AsimovConfig:
"""Local adapter configuration."""

data_root: str = ""
device: str = "cpu"


class AsimovAdapter:
"""Adapter for the Asimov embodied-AI safety benchmark."""

def __init__(self, config: AsimovConfig | None = None) -> None:
self._config = config or AsimovConfig()

@property
def adapter_id(self) -> str:
return _ADAPTER_ID

@property
def adapter_version(self) -> str:
return _ADAPTER_VERSION

async def availability(self) -> AvailabilityResult:
return command_availability(
_ADAPTER_ID, _COMMAND_ENV, _HOMEPAGE, requires_data_root=True,
)

async def list_scenarios(
self, *, tags: Sequence[str] = (), limit: int = 0,
) -> tuple[Scenario, ...]:
scenarios = load_adapter_manifest(_ADAPTER_ID)
if tags:
tag_set = set(tags)
scenarios = tuple(s for s in scenarios if tag_set.intersection(s.tags))
if limit > 0:
scenarios = scenarios[:limit]
return scenarios

async def run_trial(
self,
scenario: Scenario,
*,
seed: int = 42,
timeout_seconds: float = 300.0,
parameters: Mapping[str, Any] | None = None,
) -> TrialResult:
avail = await self.availability()
if not avail.available:
return unavailable_result(
_ADAPTER_ID, _ADAPTER_VERSION, scenario,
seed=seed, reason=avail.reason,
)

result = run_configured_command(
_COMMAND_ENV,
scenario,
seed=seed,
timeout_seconds=timeout_seconds,
)
return trial_from_subprocess(
_ADAPTER_ID, _ADAPTER_VERSION, scenario, result, seed=seed,
)
91 changes: 91 additions & 0 deletions src/benchmarks/adapters/attackvla.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,91 @@
# Copyright (c) Alibaba, Inc. and its affiliates.
"""AttackVLA adapter — adversarial robustness for vision-language-action models.

AttackVLA evaluates the robustness of VLA-based robotic policies against
adversarial perturbations applied to visual, language, or action channels.

Required: an operator-configured command from the official benchmark repository.
"""

from __future__ import annotations

from dataclasses import dataclass
from typing import Any, Mapping, Sequence

from benchmarks.adapters.base import (
command_availability,
load_adapter_manifest,
run_configured_command,
trial_from_subprocess,
unavailable_result,
)
from benchmarks.models import AvailabilityResult, Scenario, TrialResult

_ADAPTER_ID = "attackvla"
_ADAPTER_VERSION = "0.1.0"
_COMMAND_ENV = "LEAPFLOW_BENCHMARK_ATTACKVLA_COMMAND"
_REPOSITORY = "https://github.com/lijayuTnT/AttackVLA"


@dataclass(frozen=True)
class AttackVLAConfig:
"""Local adapter configuration."""

attack_type: str = "pgd"
epsilon: float = 0.03


class AttackVLAAdapter:
"""Adapter for the AttackVLA adversarial robustness benchmark."""

def __init__(self, config: AttackVLAConfig | None = None) -> None:
self._config = config or AttackVLAConfig()

@property
def adapter_id(self) -> str:
return _ADAPTER_ID

@property
def adapter_version(self) -> str:
return _ADAPTER_VERSION

async def availability(self) -> AvailabilityResult:
return command_availability(
_ADAPTER_ID, _COMMAND_ENV, _REPOSITORY, requires_data_root=True,
)

async def list_scenarios(
self, *, tags: Sequence[str] = (), limit: int = 0,
) -> tuple[Scenario, ...]:
scenarios = load_adapter_manifest(_ADAPTER_ID)
if tags:
tag_set = set(tags)
scenarios = tuple(s for s in scenarios if tag_set.intersection(s.tags))
if limit > 0:
scenarios = scenarios[:limit]
return scenarios

async def run_trial(
self,
scenario: Scenario,
*,
seed: int = 42,
timeout_seconds: float = 300.0,
parameters: Mapping[str, Any] | None = None,
) -> TrialResult:
avail = await self.availability()
if not avail.available:
return unavailable_result(
_ADAPTER_ID, _ADAPTER_VERSION, scenario,
seed=seed, reason=avail.reason,
)

result = run_configured_command(
_COMMAND_ENV,
scenario,
seed=seed,
timeout_seconds=timeout_seconds,
)
return trial_from_subprocess(
_ADAPTER_ID, _ADAPTER_VERSION, scenario, result, seed=seed,
)
Loading
Loading