diff --git a/.github/workflows/bazel.yml b/.github/workflows/bazel.yml index 64e831e5fd34..7549a4533c1d 100644 --- a/.github/workflows/bazel.yml +++ b/.github/workflows/bazel.yml @@ -20,32 +20,15 @@ jobs: strategy: fail-fast: false matrix: - include: - # macOS - - os: macos-15-xlarge - target: aarch64-apple-darwin - - os: macos-15-xlarge - target: x86_64-apple-darwin - - # Linux - - os: ubuntu-24.04 - target: x86_64-unknown-linux-gnu - - os: ubuntu-24.04 - target: x86_64-unknown-linux-musl - # 2026-02-27 Bazel tests have been flaky on arm in CI. - # Disable until we can investigate and stabilize them. - # - os: ubuntu-24.04-arm - # target: aarch64-unknown-linux-musl - # - os: ubuntu-24.04-arm - # target: aarch64-unknown-linux-gnu - - # TODO: Enable Windows once we fix the toolchain issues there. - #- os: windows-latest - # target: x86_64-pc-windows-gnullvm - runs-on: ${{ matrix.os }} + include: ${{ fromJSON(github.repository == 'openai/codex' && '[{"os":"macos-15-xlarge","fallback_os":"macos-15","target":"aarch64-apple-darwin"},{"os":"macos-15-xlarge","fallback_os":"macos-15-intel","target":"x86_64-apple-darwin"},{"os":"ubuntu-24.04","target":"x86_64-unknown-linux-gnu"},{"os":"ubuntu-24.04","target":"x86_64-unknown-linux-musl"}]' || '[{"os":"macos-15-xlarge","fallback_os":"macos-15","target":"aarch64-apple-darwin"},{"os":"ubuntu-24.04","target":"x86_64-unknown-linux-gnu"}]') }} + # macOS larger runners are available in upstream, but forks should fall back + # to the standard hosted labels that match the target architecture. Forks + # also use a reduced hosted subset for now because the hosted Intel macOS + # and Linux musl Bazel lanes are not stable in this fork's CI environment. + runs-on: ${{ github.repository == 'openai/codex' && matrix.os || matrix.fallback_os || matrix.os }} # Configure a human readable name for each job - name: Local Bazel build on ${{ matrix.os }} for ${{ matrix.target }} + name: Local Bazel build on ${{ github.repository == 'openai/codex' && matrix.os || matrix.fallback_os || matrix.os }} for ${{ matrix.target }} steps: - uses: actions/checkout@v6 diff --git a/.github/workflows/rust-ci.yml b/.github/workflows/rust-ci.yml index 46de51bc693b..adb099804e11 100644 --- a/.github/workflows/rust-ci.yml +++ b/.github/workflows/rust-ci.yml @@ -85,8 +85,10 @@ jobs: # --- CI to validate on different os/targets -------------------------------- lint_build: - name: Lint/Build — ${{ matrix.runner }} - ${{ matrix.target }}${{ matrix.profile == 'release' && ' (release)' || '' }} - runs-on: ${{ matrix.runs_on || matrix.runner }} + name: Lint/Build — ${{ github.repository == 'openai/codex' && matrix.runner || matrix.fallback_runner || matrix.runner }} - ${{ matrix.target }}${{ matrix.profile == 'release' && ' (release)' || '' }} + # Keep upstream on its larger/custom runners, but let forks run the same + # matrix on standard hosted labels they can actually provision. + runs-on: ${{ github.repository == 'openai/codex' && (matrix.upstream_runs_on || matrix.runner) || matrix.fallback_runner || matrix.runner }} timeout-minutes: 30 needs: changed # Keep job-level if to avoid spinning up runners when not needed @@ -107,45 +109,53 @@ jobs: matrix: include: - runner: macos-15-xlarge + fallback_runner: macos-15 target: aarch64-apple-darwin profile: dev - runner: macos-15-xlarge + fallback_runner: macos-15-intel target: x86_64-apple-darwin profile: dev - runner: ubuntu-24.04 + fallback_runner: ubuntu-24.04 target: x86_64-unknown-linux-musl profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-linux-x64 - runner: ubuntu-24.04 + fallback_runner: ubuntu-24.04 target: x86_64-unknown-linux-gnu profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-linux-x64 - runner: ubuntu-24.04-arm + fallback_runner: ubuntu-24.04-arm target: aarch64-unknown-linux-musl profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-linux-arm64 - runner: ubuntu-24.04-arm + fallback_runner: ubuntu-24.04-arm target: aarch64-unknown-linux-gnu profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-linux-arm64 - runner: windows-x64 + fallback_runner: windows-2025 target: x86_64-pc-windows-msvc profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-windows-x64 - runner: windows-arm64 + fallback_runner: windows-11-arm target: aarch64-pc-windows-msvc profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-windows-arm64 @@ -154,30 +164,35 @@ jobs: # Hopefully this also pre-populates the build cache to speed up # releases. - runner: macos-15-xlarge + fallback_runner: macos-15 target: aarch64-apple-darwin profile: release - runner: ubuntu-24.04 + fallback_runner: ubuntu-24.04 target: x86_64-unknown-linux-musl profile: release - runs_on: + upstream_runs_on: group: codex-runners labels: codex-linux-x64 - runner: ubuntu-24.04-arm + fallback_runner: ubuntu-24.04-arm target: aarch64-unknown-linux-musl profile: release - runs_on: + upstream_runs_on: group: codex-runners labels: codex-linux-arm64 - runner: windows-x64 + fallback_runner: windows-2025 target: x86_64-pc-windows-msvc profile: release - runs_on: + upstream_runs_on: group: codex-runners labels: codex-windows-x64 - runner: windows-arm64 + fallback_runner: windows-11-arm target: aarch64-pc-windows-msvc profile: release - runs_on: + upstream_runs_on: group: codex-runners labels: codex-windows-arm64 @@ -451,8 +466,8 @@ jobs: key: apt-${{ matrix.runner }}-${{ matrix.target }}-v1 tests: - name: Tests — ${{ matrix.runner }} - ${{ matrix.target }} - runs-on: ${{ matrix.runs_on || matrix.runner }} + name: Tests — ${{ github.repository == 'openai/codex' && matrix.runner || matrix.fallback_runner || matrix.runner }} - ${{ matrix.target }} + runs-on: ${{ github.repository == 'openai/codex' && (matrix.upstream_runs_on || matrix.runner) || matrix.fallback_runner || matrix.runner }} timeout-minutes: 30 needs: changed if: ${{ needs.changed.outputs.codex == 'true' || needs.changed.outputs.workflows == 'true' || github.event_name == 'push' }} @@ -470,39 +485,51 @@ jobs: matrix: include: - runner: macos-15-xlarge + fallback_runner: macos-15 target: aarch64-apple-darwin profile: dev - runner: ubuntu-24.04 + fallback_runner: ubuntu-24.04 target: x86_64-unknown-linux-gnu profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-linux-x64 - runner: ubuntu-24.04-arm + fallback_runner: ubuntu-24.04-arm target: aarch64-unknown-linux-gnu profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-linux-arm64 - runner: windows-x64 + fallback_runner: windows-2025 target: x86_64-pc-windows-msvc profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-windows-x64 - runner: windows-arm64 + fallback_runner: windows-11-arm target: aarch64-pc-windows-msvc profile: dev - runs_on: + upstream_runs_on: group: codex-runners labels: codex-windows-arm64 steps: - uses: actions/checkout@v6 - name: Set up Node.js for js_repl tests + if: ${{ matrix.target != 'aarch64-pc-windows-msvc' }} uses: actions/setup-node@v6 with: node-version-file: codex-rs/node-version.txt + - name: Set up x64 Node.js for js_repl tests (Windows ARM) + if: ${{ matrix.target == 'aarch64-pc-windows-msvc' }} + uses: actions/setup-node@v6 + with: + node-version-file: codex-rs/node-version.txt + architecture: x64 - name: Install Linux build dependencies if: ${{ runner.os == 'Linux' }} shell: bash @@ -596,6 +623,11 @@ jobs: sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0 fi + - name: Prebuild CLI test binaries + run: | + cargo build --all-features -p codex-cli --bin codex --target ${{ matrix.target }} --profile ci-test + cargo build --all-features -p codex-rmcp-client --bin test_stdio_server --target ${{ matrix.target }} --profile ci-test + - name: tests id: test run: cargo nextest run --all-features --no-fail-fast --target ${{ matrix.target }} --cargo-profile ci-test --timings diff --git a/.prettierignore b/.prettierignore index f5b50f6ba228..03fce3713df2 100644 --- a/.prettierignore +++ b/.prettierignore @@ -5,3 +5,6 @@ pnpm-lock.yaml prompt.md *_prompt.md *_instructions.md + +# Vendored parser bundle; keep upstream minified formatting intact. +codex-rs/core/src/tools/js_repl/meriyah.umd.min.js diff --git a/codex-rs/app-server/tests/common/models_cache.rs b/codex-rs/app-server/tests/common/models_cache.rs index 62a8f55d1bac..53b5bc5f6f9b 100644 --- a/codex-rs/app-server/tests/common/models_cache.rs +++ b/codex-rs/app-server/tests/common/models_cache.rs @@ -50,6 +50,10 @@ fn preset_to_info(preset: &ModelPreset, priority: i32) -> ModelInfo { } } +fn default_openai_provider_scope() -> String { + "provider_name=OpenAI;base_url=https://api.openai.com/v1;auth_mode=None".to_string() +} + /// Write a models_cache.json file to the codex home directory. /// This prevents ModelsManager from making network requests to refresh models. /// The cache will be treated as fresh (within TTL) and used instead of fetching from the network. @@ -89,6 +93,7 @@ pub fn write_models_cache_with_models( "fetched_at": fetched_at, "etag": null, "client_version": client_version, + "provider_scope": default_openai_provider_scope(), "models": models }); std::fs::write(cache_path, serde_json::to_string_pretty(&cache)?) diff --git a/codex-rs/cli/build.rs b/codex-rs/cli/build.rs index 8b40000c4c48..39b8da32cbcb 100644 --- a/codex-rs/cli/build.rs +++ b/codex-rs/cli/build.rs @@ -4,7 +4,10 @@ use std::path::PathBuf; use std::process::Command; fn main() { - let manifest_dir = std::env::var("CARGO_MANIFEST_DIR").expect("CARGO_MANIFEST_DIR is set"); + let Some(manifest_dir) = std::env::var_os("CARGO_MANIFEST_DIR") else { + println!("cargo:rustc-env=CODEX_CLI_VERSION=unknown"); + return; + }; let repo_root = Path::new(&manifest_dir).join("../.."); configure_git_rerun_inputs(&repo_root); diff --git a/codex-rs/core/src/codex.rs b/codex-rs/core/src/codex.rs index 44a3815a9bd3..849a0437a0d1 100644 --- a/codex-rs/core/src/codex.rs +++ b/codex-rs/core/src/codex.rs @@ -5669,7 +5669,10 @@ pub(crate) async fn run_turn( // Run reflection if enabled and we haven't exceeded max attempts. let max_attempts = reflection_config.max_attempts; - if reflection_enabled && reflection_attempt < max_attempts { + if reflection_enabled + && reflection_attempt < max_attempts + && let Some(reflection_model_info) = reflection_model_info.as_ref() + { reflection_attempt += 1; info!( "Running reflection evaluation (attempt {}/{})", @@ -5687,9 +5690,6 @@ pub(crate) async fn run_turn( max_attempts, ); - let reflection_model_info = reflection_model_info - .as_ref() - .expect("reflection model info"); match evaluate_reflection( &sess.services.model_client, reflection_model_info, diff --git a/codex-rs/core/src/memories/prompts.rs b/codex-rs/core/src/memories/prompts.rs index 35cfe1edf07b..6c3a76e1928e 100644 --- a/codex-rs/core/src/memories/prompts.rs +++ b/codex-rs/core/src/memories/prompts.rs @@ -3,7 +3,6 @@ use crate::memories::phase_one; use crate::memories::storage::rollout_summary_file_stem_from_parts; use crate::truncate::TruncationPolicy; use crate::truncate::truncate_text; -use askama::Template; use codex_protocol::openai_models::ModelInfo; use codex_state::Phase2InputSelection; use codex_state::Stage1Output; @@ -12,27 +11,14 @@ use std::path::Path; use tokio::fs; use tracing::warn; -#[derive(Template)] -#[template(path = "memories/consolidation.md", escape = "none")] -struct ConsolidationPromptTemplate<'a> { - memory_root: &'a str, - phase2_input_selection: &'a str, -} - -#[derive(Template)] -#[template(path = "memories/stage_one_input.md", escape = "none")] -struct StageOneInputTemplate<'a> { - rollout_path: &'a str, - rollout_cwd: &'a str, - rollout_contents: &'a str, -} +#[allow(dead_code)] +type AskamaDependencyMarker = askama::Error; -#[derive(Template)] -#[template(path = "memories/read_path.md", escape = "none")] -struct MemoryToolDeveloperInstructionsTemplate<'a> { - base_path: &'a str, - memory_summary: &'a str, -} +const CONSOLIDATION_PROMPT_TEMPLATE: &str = + include_str!("../../templates/memories/consolidation.md"); +const STAGE_ONE_INPUT_TEMPLATE: &str = include_str!("../../templates/memories/stage_one_input.md"); +const MEMORY_TOOL_DEVELOPER_INSTRUCTIONS_TEMPLATE: &str = + include_str!("../../templates/memories/read_path.md"); /// Builds the consolidation subagent prompt for a specific memory root. pub(super) fn build_consolidation_prompt( @@ -41,16 +27,23 @@ pub(super) fn build_consolidation_prompt( ) -> String { let memory_root = memory_root.display().to_string(); let phase2_input_selection = render_phase2_input_selection(selection); - let template = ConsolidationPromptTemplate { - memory_root: &memory_root, - phase2_input_selection: &phase2_input_selection, - }; - template.render().unwrap_or_else(|err| { - warn!("failed to render memories consolidation prompt template: {err}"); - format!( + let rendered = render_template( + CONSOLIDATION_PROMPT_TEMPLATE, + &[ + ("{{ memory_root }}", memory_root.as_str()), + ( + "{{ phase2_input_selection }}", + phase2_input_selection.as_str(), + ), + ], + ); + if rendered.contains("{{ ") { + warn!("failed to fully render memories consolidation prompt template"); + return format!( "## Memory Phase 2 (Consolidation)\nConsolidate Codex memories in: {memory_root}\n\n{phase2_input_selection}" - ) - }) + ); + } + rendered } fn render_phase2_input_selection(selection: &Phase2InputSelection) -> String { @@ -144,12 +137,17 @@ pub(super) fn build_stage_one_input_message( let rollout_path = rollout_path.display().to_string(); let rollout_cwd = rollout_cwd.display().to_string(); - Ok(StageOneInputTemplate { - rollout_path: &rollout_path, - rollout_cwd: &rollout_cwd, - rollout_contents: &truncated_rollout_contents, - } - .render()?) + Ok(render_template( + STAGE_ONE_INPUT_TEMPLATE, + &[ + ("{{ rollout_path }}", rollout_path.as_str()), + ("{{ rollout_cwd }}", rollout_cwd.as_str()), + ( + "{{ rollout_contents }}", + truncated_rollout_contents.as_str(), + ), + ], + )) } /// Build prompt used for read path. This prompt must be added to the developer instructions. In @@ -171,11 +169,21 @@ pub(crate) async fn build_memory_tool_developer_instructions(codex_home: &Path) return None; } let base_path = base_path.display().to_string(); - let template = MemoryToolDeveloperInstructionsTemplate { - base_path: &base_path, - memory_summary: &memory_summary, - }; - template.render().ok() + Some(render_template( + MEMORY_TOOL_DEVELOPER_INSTRUCTIONS_TEMPLATE, + &[ + ("{{ base_path }}", base_path.as_str()), + ("{{ memory_summary }}", memory_summary.as_str()), + ], + )) +} + +fn render_template(template: &str, replacements: &[(&str, &str)]) -> String { + replacements + .iter() + .fold(template.to_string(), |rendered, (needle, value)| { + rendered.replace(needle, value) + }) } #[cfg(test)] diff --git a/codex-rs/core/src/models_manager/manager.rs b/codex-rs/core/src/models_manager/manager.rs index 08dff42f4942..cb0089beeac7 100644 --- a/codex-rs/core/src/models_manager/manager.rs +++ b/codex-rs/core/src/models_manager/manager.rs @@ -586,7 +586,7 @@ impl ModelsManager { let mut models = payload .data .into_iter() - .filter(|model| model.is_picker_enabled()) + .filter(OpenAiCompatModel::is_picker_enabled) .collect::>(); models.sort_by_key(|model| !model.supports_responses_endpoint()); let model_ids = models.into_iter().map(|model| model.id).collect::>(); @@ -649,7 +649,7 @@ impl ModelsManager { let model_ids = payload .models .into_iter() - .filter(|model| model.is_picker_enabled()) + .filter(OllamaTagsModel::is_picker_enabled) .map(|model| model.name) .collect::>(); let models = self.map_provider_model_ids(model_ids); @@ -842,7 +842,7 @@ impl ModelsManager { fn extract_host_from_url(input: &str) -> Option { reqwest::Url::parse(input) .ok() - .and_then(|url| url.host_str().map(|host| host.to_ascii_lowercase())) + .and_then(|url| url.host_str().map(str::to_ascii_lowercase)) } fn ollama_tags_url(base_url: &str) -> String { diff --git a/codex-rs/core/src/tools/js_repl/kernel.js b/codex-rs/core/src/tools/js_repl/kernel.js index e8f0ac937ded..f87bb529f62c 100644 --- a/codex-rs/core/src/tools/js_repl/kernel.js +++ b/codex-rs/core/src/tools/js_repl/kernel.js @@ -9,9 +9,12 @@ const { builtinModules, createRequire } = require("node:module"); const { createInterface } = require("node:readline"); const { performance } = require("node:perf_hooks"); const path = require("node:path"); -const { URL, URLSearchParams, fileURLToPath, pathToFileURL } = require( - "node:url", -); +const { + URL, + URLSearchParams, + fileURLToPath, + pathToFileURL, +} = require("node:url"); const { inspect, TextDecoder, TextEncoder } = require("node:util"); const vm = require("node:vm"); @@ -318,7 +321,9 @@ function resolvePathSpecifier(specifier, referrerIdentifier = null) { try { candidate = fileURLToPath(new URL(specifier)); } catch (err) { - throw new Error(`Failed to resolve module "${specifier}": ${err.message}`); + throw new Error( + `Failed to resolve module "${specifier}": ${err.message}`, + ); } } else { const baseDir = @@ -440,14 +445,19 @@ async function loadLinkedFileModule(modulePath) { setImportMeta(meta, mod, false); }, importModuleDynamically(specifier, referrer) { - return importResolved(resolveSpecifier(specifier, referrer?.identifier)); + return importResolved( + resolveSpecifier(specifier, referrer?.identifier), + ); }, }); linkedFileModules.set(modulePath, module); } if (module.status === "unlinked") { await module.link(async (specifier, referencingModule) => { - const resolved = resolveSpecifier(specifier, referencingModule?.identifier); + const resolved = resolveSpecifier( + specifier, + referencingModule?.identifier, + ); if (resolved.kind !== "file") { throw new Error( `Static import "${specifier}" is not supported from js_repl local files. Use await import("${specifier}") instead.`, @@ -599,7 +609,10 @@ function instrumentVariableDeclarationSource( return code.slice(declaration.start, declaration.end); } - const prefix = code.slice(declaration.start, declaration.declarations[0].start); + const prefix = code.slice( + declaration.start, + declaration.declarations[0].start, + ); const suffix = code.slice( declaration.declarations[declaration.declarations.length - 1].end, declaration.end, @@ -699,10 +712,7 @@ function collectHoistedVarDeclarationStarts(ast) { function collectFutureVarWriteReplacements( code, ast, - { - helperDeclarations = null, - markCommittedFnName = null, - } = {}, + { helperDeclarations = null, markCommittedFnName = null } = {}, ) { // Failed-cell hoisted tracking intentionally stays small here. We only mark // direct top-level writes to future `var` bindings, plus top-level @@ -790,11 +800,7 @@ function collectFutureVarWriteReplacements( const helperName = nextInternalBindingName(); helperDeclarations.push(`let ${helperName};`); const shortCircuitOperator = - node.operator === "&&=" - ? "&&" - : node.operator === "||=" - ? "||" - : "??"; + node.operator === "&&=" ? "&&" : node.operator === "||=" ? "||" : "??"; addReplacement( node.start, node.end, @@ -1045,11 +1051,7 @@ async function buildModuleSource(code) { } function canReadCommittedBinding(module, binding) { - if ( - !module || - binding.kind === "var" || - binding.kind === "function" - ) { + if (!module || binding.kind === "var" || binding.kind === "function") { return false; } @@ -1321,7 +1323,9 @@ function normalizeMcpImageData(data, mimeType) { return data; } const normalizedMimeType = - typeof mimeType === "string" && mimeType ? mimeType : "application/octet-stream"; + typeof mimeType === "string" && mimeType + ? mimeType + : "application/octet-stream"; return `data:${normalizedMimeType};base64,${data}`; } @@ -1336,7 +1340,10 @@ function parseMcpToolResult(result) { if ("Err" in result) { const error = result.Err; - return { images: [], textCount: typeof error === "string" && error ? 1 : 0 }; + return { + images: [], + textCount: typeof error === "string" && error ? 1 : 0, + }; } if (!("Ok" in result)) { @@ -1356,7 +1363,10 @@ function parseMcpToolResult(result) { } if (item.type === "image") { images.push({ - image_url: normalizeMcpImageData(item.data, item.mimeType ?? item.mime_type), + image_url: normalizeMcpImageData( + item.data, + item.mimeType ?? item.mime_type, + ), }); continue; } @@ -1374,7 +1384,9 @@ function parseMcpToolResult(result) { function requireSingleImage(parsed) { if (parsed.textCount > 0) { - throw new Error("codex.emitImage does not accept mixed text and image content"); + throw new Error( + "codex.emitImage does not accept mixed text and image content", + ); } if (parsed.images.length !== 1) { throw new Error("codex.emitImage expected exactly one image"); @@ -1556,7 +1568,9 @@ async function handleExec(message) { meta.__codexInternalMarkPreludeCompleted = markPreludeCompleted; }, importModuleDynamically(specifier, referrer) { - return importResolved(resolveSpecifier(specifier, referrer?.identifier)); + return importResolved( + resolveSpecifier(specifier, referrer?.identifier), + ); }, }); @@ -1587,7 +1601,9 @@ async function handleExec(message) { await module.evaluate(); if (pendingBackgroundTasks.size > 0) { - const backgroundResults = await Promise.all([...pendingBackgroundTasks]); + const backgroundResults = await Promise.all([ + ...pendingBackgroundTasks, + ]); const firstUnhandledBackgroundError = backgroundResults.find( (result) => !result.ok && !result.observation.observed, ); @@ -1611,11 +1627,11 @@ async function handleExec(message) { } catch (error) { const { bindings: committedBindings, committedCurrentBindingCount } = collectCommittedBindings( - moduleLinked ? module : null, - priorBindings, - currentBindings, - committedCurrentBindingNames, - ); + moduleLinked ? module : null, + priorBindings, + currentBindings, + committedCurrentBindingNames, + ); // Preserve the last successfully linked module across link-time failures. // A module whose link step failed cannot safely back @prev because reading // its namespace throws before evaluation ever begins. Likewise, if a diff --git a/codex-rs/core/tests/common/lib.rs b/codex-rs/core/tests/common/lib.rs index a75a79843871..416b710218cb 100644 --- a/codex-rs/core/tests/common/lib.rs +++ b/codex-rs/core/tests/common/lib.rs @@ -15,7 +15,6 @@ use codex_protocol::openai_models::ModelVisibility; use codex_protocol::openai_models::ModelsResponse; use codex_utils_absolute_path::AbsolutePathBuf; use regex_lite::Regex; -use serde_json; use std::path::PathBuf; pub mod apps_test_server; @@ -159,10 +158,26 @@ pub async fn load_default_config_for_test(codex_home: &TempDir) -> Config { config } +pub fn is_default_test_model_catalog(model_catalog: &ModelsResponse) -> bool { + *model_catalog == build_test_model_catalog(None) +} + +fn load_bundled_models_response() -> anyhow::Result { + let bundled_models_path = codex_utils_cargo_bin::find_resource!("../../models.json") + .context("bundled models.json")?; + let bundled_models_contents = + std::fs::read_to_string(&bundled_models_path).with_context(|| { + format!( + "read bundled models.json from {}", + bundled_models_path.display() + ) + })?; + serde_json::from_str(&bundled_models_contents).context("parse bundled models.json") +} + fn build_test_model_catalog(existing: Option) -> ModelsResponse { let mut response = existing.unwrap_or_else(|| { - serde_json::from_str(include_str!("../../models.json")) - .expect("bundled models.json should parse") + load_bundled_models_response().expect("bundled models.json should load") }); add_test_model( @@ -333,7 +348,49 @@ pub fn format_with_current_shell_display_non_login(command: &str) -> String { } pub fn stdio_server_bin() -> Result { - codex_utils_cargo_bin::cargo_bin("test_stdio_server").map(|p| p.to_string_lossy().to_string()) + workspace_binary("codex-rmcp-client", "test_stdio_server") + .map(|p| p.to_string_lossy().to_string()) +} + +fn workspace_binary(package: &str, binary: &str) -> Result { + if let Some(path) = sibling_test_binary(binary) { + return Ok(path); + } + + match codex_utils_cargo_bin::cargo_bin(binary) { + Ok(path) => Ok(path), + Err(original_err) => { + if codex_utils_cargo_bin::runfiles_available() { + return Err(original_err); + } + + let Ok(repo_root) = codex_utils_cargo_bin::repo_root() else { + return Err(original_err); + }; + let cargo = std::env::var_os("CARGO").unwrap_or_else(|| "cargo".into()); + let status = std::process::Command::new(cargo) + .current_dir(repo_root.join("codex-rs")) + .arg("build") + .arg("-p") + .arg(package) + .arg("--bin") + .arg(binary) + .status(); + match status { + Ok(status) if status.success() => { + codex_utils_cargo_bin::cargo_bin(binary).map_err(|_| original_err) + } + _ => Err(original_err), + } + } + } +} + +fn sibling_test_binary(binary: &str) -> Option { + let current_exe = std::env::current_exe().ok()?; + let profile_dir = current_exe.parent()?.parent()?; + let candidate = profile_dir.join(format!("{binary}{}", std::env::consts::EXE_SUFFIX)); + candidate.is_file().then_some(candidate) } pub mod fs_wait { diff --git a/codex-rs/core/tests/common/test_codex.rs b/codex-rs/core/tests/common/test_codex.rs index 9730f6a45445..89929beb47e9 100644 --- a/codex-rs/core/tests/common/test_codex.rs +++ b/codex-rs/core/tests/common/test_codex.rs @@ -3,7 +3,6 @@ use std::path::Path; use std::path::PathBuf; use std::sync::Arc; -use anyhow::Context; use anyhow::Result; use codex_core::CodexAuth; use codex_core::CodexThread; @@ -179,12 +178,16 @@ impl TestCodexBuilder { resume_from: Option, ) -> anyhow::Result { let auth = self.auth.clone(); - let thread_manager = if let Some(model_catalog) = config.model_catalog.clone() { + let use_authoritative_model_catalog = config + .model_catalog + .as_ref() + .is_some_and(|model_catalog| !crate::is_default_test_model_catalog(model_catalog)); + let thread_manager = if use_authoritative_model_catalog { ThreadManager::new( config.codex_home.clone(), codex_core::test_support::auth_manager_from_auth(auth.clone()), SessionSource::Exec, - Some(model_catalog), + config.model_catalog.clone(), CollaborationModesConfig::default(), config.model_provider.clone(), ) @@ -272,17 +275,7 @@ fn ensure_test_model_catalog(config: &mut Config) -> Result<()> { return Ok(()); } - let bundled_models_path = codex_utils_cargo_bin::find_resource!("../../models.json") - .context("bundled models.json")?; - let bundled_models_contents = - std::fs::read_to_string(&bundled_models_path).with_context(|| { - format!( - "read bundled models.json from {}", - bundled_models_path.display() - ) - })?; - let bundled_models: ModelsResponse = - serde_json::from_str(&bundled_models_contents).context("parse bundled models.json")?; + let bundled_models = super::load_bundled_models_response()?; let mut model = bundled_models .models .iter() @@ -559,8 +552,12 @@ pub fn test_codex() -> TestCodexBuilder { #[cfg(test)] mod tests { use super::*; + use crate::responses::mount_models_once; + use codex_core::models_manager::manager::RefreshStrategy; + use codex_protocol::openai_models::ModelsResponse; use pretty_assertions::assert_eq; use serde_json::json; + use wiremock::MockServer; #[test] fn custom_tool_call_output_text_returns_output_text() { @@ -587,4 +584,115 @@ mod tests { let _ = custom_tool_call_output_text(&bodies, "call-2"); } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn default_test_model_catalog_still_allows_remote_model_refresh() { + let server = MockServer::start().await; + let mut remote_catalog = + crate::load_bundled_models_response().expect("valid bundled models.json"); + let remote_slug = "test-codex-builder-remote"; + let remote_model = remote_catalog + .models + .iter_mut() + .find(|model| model.slug == "gpt-5.1") + .expect("gpt-5.1 should exist in bundled models.json"); + remote_model.slug = remote_slug.to_string(); + remote_model.display_name = remote_slug.to_string(); + let models_mock = mount_models_once( + &server, + ModelsResponse { + models: vec![remote_model.clone()], + }, + ) + .await; + + let mut builder = + test_codex().with_auth(CodexAuth::create_dummy_chatgpt_auth_for_testing()); + let test = builder + .build(&server) + .await + .expect("build test codex with remote provider"); + + let available = test + .thread_manager + .get_models_manager() + .list_models(RefreshStrategy::OnlineIfUncached) + .await; + + assert!( + available.iter().any(|model| model.model == remote_slug), + "default test catalog should not block remote models refresh: {available:?}" + ); + assert_eq!( + models_mock.requests().len(), + 1, + "default test catalog should still hit /models exactly once" + ); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn explicit_model_catalog_remains_authoritative() { + let server = MockServer::start().await; + let mut remote_catalog = + crate::load_bundled_models_response().expect("valid bundled models.json"); + let remote_slug = "test-codex-builder-remote"; + let remote_model = remote_catalog + .models + .iter_mut() + .find(|model| model.slug == "gpt-5.1") + .expect("gpt-5.1 should exist in bundled models.json"); + remote_model.slug = remote_slug.to_string(); + remote_model.display_name = remote_slug.to_string(); + let models_mock = mount_models_once( + &server, + ModelsResponse { + models: vec![remote_model.clone()], + }, + ) + .await; + + let mut custom_catalog = + crate::load_bundled_models_response().expect("valid bundled models.json"); + custom_catalog.models.truncate(1); + let custom_model = custom_catalog + .models + .first_mut() + .expect("custom catalog should include one model"); + custom_model.slug = "custom-catalog-only".to_string(); + custom_model.display_name = "custom-catalog-only".to_string(); + + let custom_catalog_for_config = custom_catalog.clone(); + let mut builder = test_codex() + .with_auth(CodexAuth::create_dummy_chatgpt_auth_for_testing()) + .with_config(move |config| { + config.model = Some("custom-catalog-only".to_string()); + config.model_catalog = Some(custom_catalog_for_config); + }); + let test = builder + .build(&server) + .await + .expect("build test codex with custom catalog"); + + let available = test + .thread_manager + .get_models_manager() + .list_models(RefreshStrategy::OnlineIfUncached) + .await; + + assert!( + available + .iter() + .any(|model| model.model == "custom-catalog-only"), + "explicit custom catalog should stay available: {available:?}" + ); + assert!( + !available.iter().any(|model| model.model == remote_slug), + "explicit custom catalog should ignore remote /models refresh: {available:?}" + ); + assert_eq!( + models_mock.requests().len(), + 0, + "explicit custom catalog should not fetch /models" + ); + } } diff --git a/codex-rs/core/tests/suite/cli_stream.rs b/codex-rs/core/tests/suite/cli_stream.rs index 0ed722f79150..5039e15c81cc 100644 --- a/codex-rs/core/tests/suite/cli_stream.rs +++ b/codex-rs/core/tests/suite/cli_stream.rs @@ -643,12 +643,8 @@ async fn cli_copilot_provider_models_request_omits_openai_intent_header() { // Verify a /responses request was made, proving the Gemini model was // accepted and used for the prompt. - let responses_requests: Vec<_> = requests - .iter() - .filter(|r| r.url.path().contains("/responses")) - .collect(); assert!( - !responses_requests.is_empty(), + requests.iter().any(|r| r.url.path().contains("/responses")), "expected a POST /responses request, proving gemini-2.5-pro was \ accepted as a valid model from the copilot provider" ); diff --git a/codex-rs/core/tests/suite/eval_swe_bench.rs b/codex-rs/core/tests/suite/eval_swe_bench.rs index 75307e026051..e49ee28b6156 100644 --- a/codex-rs/core/tests/suite/eval_swe_bench.rs +++ b/codex-rs/core/tests/suite/eval_swe_bench.rs @@ -12,7 +12,6 @@ //! cargo test -p codex-core --test all eval_swe -- --ignored --nocapture //! ``` -use assert_cmd::prelude::*; use std::fs; use std::io::Read; use std::io::Write; @@ -21,6 +20,7 @@ use std::process::Stdio; use std::thread; use tempfile::TempDir; +#[expect(clippy::expect_used)] fn require_azure_credentials() -> (String, String, String) { let api_key = std::env::var("AZURE_OPENAI_API_KEY").expect("AZURE_OPENAI_API_KEY env var not set"); @@ -89,6 +89,7 @@ fn run_eval_task( reflection_enabled: bool, ) -> EvalResult { #![expect(clippy::unwrap_used)] + #![expect(clippy::expect_used)] let (api_key, base_url, model) = require_azure_credentials(); @@ -114,7 +115,7 @@ fn run_eval_task( .unwrap(); // Run codex - let mut cmd = Command::cargo_bin("codex").unwrap(); + let mut cmd = Command::new(codex_utils_cargo_bin::cargo_bin("codex").unwrap()); cmd.current_dir(work_dir); cmd.env("AZURE_OPENAI_API_KEY", api_key); cmd.env("CODEX_HOME", &codex_home); diff --git a/codex-rs/core/tests/suite/models_cache_ttl.rs b/codex-rs/core/tests/suite/models_cache_ttl.rs index e8f9cbf7fef2..3ab344a70c82 100644 --- a/codex-rs/core/tests/suite/models_cache_ttl.rs +++ b/codex-rs/core/tests/suite/models_cache_ttl.rs @@ -137,7 +137,9 @@ async fn renews_cache_ttl_on_matching_models_etag() -> Result<()> { #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn uses_cache_when_version_matches() -> Result<()> { let server = MockServer::start().await; + let base_url = format!("{}/v1", server.uri()); let cached_model = test_remote_model(VERSIONED_MODEL, 1); + let provider_scope = openai_chatgpt_provider_scope(&base_url); let models_mock = responses::mount_models_once( &server, ModelsResponse { @@ -153,6 +155,7 @@ async fn uses_cache_when_version_matches() -> Result<()> { fetched_at: Utc::now(), etag: None, client_version: Some(codex_core::models_manager::client_version_to_whole()), + provider_scope: Some(provider_scope), models: vec![cached_model], }; let cache_path = home.join(CACHE_FILE); @@ -184,7 +187,9 @@ async fn uses_cache_when_version_matches() -> Result<()> { #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn refreshes_when_cache_version_missing() -> Result<()> { let server = MockServer::start().await; + let base_url = format!("{}/v1", server.uri()); let cached_model = test_remote_model(MISSING_VERSION_MODEL, 1); + let provider_scope = openai_chatgpt_provider_scope(&base_url); let models_mock = responses::mount_models_once( &server, ModelsResponse { @@ -200,6 +205,7 @@ async fn refreshes_when_cache_version_missing() -> Result<()> { fetched_at: Utc::now(), etag: None, client_version: None, + provider_scope: Some(provider_scope), models: vec![cached_model], }; let cache_path = home.join(CACHE_FILE); @@ -231,7 +237,9 @@ async fn refreshes_when_cache_version_missing() -> Result<()> { #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn refreshes_when_cache_version_differs() -> Result<()> { let server = MockServer::start().await; + let base_url = format!("{}/v1", server.uri()); let cached_model = test_remote_model(DIFFERENT_VERSION_MODEL, 1); + let provider_scope = openai_chatgpt_provider_scope(&base_url); let models_response = ModelsResponse { models: vec![test_remote_model("remote-different", 2)], }; @@ -248,6 +256,7 @@ async fn refreshes_when_cache_version_differs() -> Result<()> { fetched_at: Utc::now(), etag: None, client_version: Some(format!("{client_version}-diff")), + provider_scope: Some(provider_scope), models: vec![cached_model], }; let cache_path = home.join(CACHE_FILE); @@ -303,6 +312,10 @@ fn write_cache_sync(path: &Path, cache: &ModelsCache) -> Result<()> { Ok(()) } +fn openai_chatgpt_provider_scope(base_url: &str) -> String { + format!("provider_name=OpenAI;base_url={base_url};auth_mode=Some(Chatgpt)") +} + #[derive(Debug, Clone, Serialize, Deserialize)] struct ModelsCache { fetched_at: DateTime, @@ -310,6 +323,8 @@ struct ModelsCache { etag: Option, #[serde(default)] client_version: Option, + #[serde(default)] + provider_scope: Option, models: Vec, } diff --git a/codex-rs/core/tests/suite/reflection.rs b/codex-rs/core/tests/suite/reflection.rs index ec8ffe9e0a27..b2df8d544376 100644 --- a/codex-rs/core/tests/suite/reflection.rs +++ b/codex-rs/core/tests/suite/reflection.rs @@ -21,6 +21,7 @@ use std::process::Stdio; use std::thread; use tempfile::TempDir; +#[expect(clippy::expect_used)] fn require_azure_credentials() -> (String, String, String) { let api_key = std::env::var("AZURE_OPENAI_API_KEY") .expect("AZURE_OPENAI_API_KEY env var not set — skip running Azure tests"); @@ -73,6 +74,7 @@ fn run_azure_reflection_test( prompt: &str, ) -> (assert_cmd::assert::Assert, TempDir, Vec, Vec) { #![expect(clippy::unwrap_used)] + #![expect(clippy::expect_used)] let (api_key, base_url, model) = require_azure_credentials(); @@ -87,7 +89,7 @@ fn run_azure_reflection_test( ) .unwrap(); - let mut cmd = Command::cargo_bin("codex").unwrap(); + let mut cmd = Command::new(codex_utils_cargo_bin::cargo_bin("codex").unwrap()); cmd.current_dir(dir.path()); cmd.env("AZURE_OPENAI_API_KEY", api_key); cmd.env("CODEX_HOME", &codex_home); diff --git a/codex-rs/linux-sandbox/tests/suite/landlock.rs b/codex-rs/linux-sandbox/tests/suite/landlock.rs index 8e2f5ef41ef7..72a66b59dcd2 100644 --- a/codex-rs/linux-sandbox/tests/suite/landlock.rs +++ b/codex-rs/linux-sandbox/tests/suite/landlock.rs @@ -148,6 +148,8 @@ fn is_bwrap_unavailable_output(output: &codex_core::exec::ExecToolCallOutput) -> && (output.stderr.text.contains("Operation not permitted") || output.stderr.text.contains("Permission denied") || output.stderr.text.contains("Invalid argument"))) + || (output.stderr.text.contains("setting up uid map") + && output.stderr.text.contains("Permission denied")) } async fn should_skip_bwrap_tests() -> bool { diff --git a/codex-rs/tui/src/chatwidget/snapshots/codex_tui__chatwidget__tests__experimental_popup_reflection_runtime_enabled_linux.snap b/codex-rs/tui/src/chatwidget/snapshots/codex_tui__chatwidget__tests__experimental_popup_reflection_runtime_enabled_linux.snap new file mode 100644 index 000000000000..8652caf0ab66 --- /dev/null +++ b/codex-rs/tui/src/chatwidget/snapshots/codex_tui__chatwidget__tests__experimental_popup_reflection_runtime_enabled_linux.snap @@ -0,0 +1,21 @@ +--- +source: tui/src/chatwidget/tests.rs +expression: popup +--- + Experimental features + Toggle experimental features. Changes are saved to config.toml. + +› [ ] JavaScript REPL Enable a persistent Node-backed JavaScript REPL for interactive website debugging + and other inline JavaScript execution capabilities. Requires Node >= v22.22.0 + installed. + [x] Reflection Run a reflection judge pass before Codex finalizes a response. + [ ] Bubblewrap sandbox Try the new linux sandbox based on bubblewrap. + [ ] Multi-agents Ask Codex to spawn multiple agents to parallelize the work and win in efficiency. + [ ] Apps Use a connected ChatGPT App using "$". Install Apps via /apps command. Restart + Codex after enabling. + [ ] Automatic approval review Dispatch `on-request` approval prompts (for e.g. sandbox escapes or blocked network + access) to a carefully-prompted security reviewer subagent rather than blocking the + agent on your input. + [ ] Prevent sleep while running Keep your computer awake while Codex is running a thread. + + Press space to select or enter to save for next conversation diff --git a/codex-rs/tui/src/chatwidget/tests.rs b/codex-rs/tui/src/chatwidget/tests.rs index 1253d23b84a7..df5612888e16 100644 --- a/codex-rs/tui/src/chatwidget/tests.rs +++ b/codex-rs/tui/src/chatwidget/tests.rs @@ -7095,7 +7095,12 @@ async fn experimental_popup_marks_reflection_enabled_when_feature_flag_is_enable chat.open_experimental_popup(); let popup = render_bottom_popup(&chat, 120); - assert_snapshot!("experimental_popup_reflection_runtime_enabled", popup); + let snapshot_name = if cfg!(target_os = "linux") { + "experimental_popup_reflection_runtime_enabled_linux" + } else { + "experimental_popup_reflection_runtime_enabled" + }; + assert_snapshot!(snapshot_name, popup); } #[tokio::test] diff --git a/codex-rs/tui/tests/suite/model_availability_nux.rs b/codex-rs/tui/tests/suite/model_availability_nux.rs index 512db979748c..f447ee6496b6 100644 --- a/codex-rs/tui/tests/suite/model_availability_nux.rs +++ b/codex-rs/tui/tests/suite/model_availability_nux.rs @@ -53,14 +53,17 @@ async fn resume_startup_does_not_consume_model_availability_nux_count() -> Resul serde_json::to_string(&source_catalog)?, )?; - let repo_root_display = repo_root.display(); - let catalog_display = custom_catalog_path.display(); + let repo_root_display = repo_root.display().to_string(); + let repo_root_toml = toml::Value::String(repo_root_display).to_string(); + let catalog_display = custom_catalog_path.display().to_string(); + let catalog_toml = toml::Value::String(catalog_display).to_string(); + let model_slug_toml = toml::Value::String(model_slug.clone()).to_string(); let config_contents = format!( - r#"model = "{model_slug}" + r#"model = {model_slug_toml} model_provider = "openai" -model_catalog_json = "{catalog_display}" +model_catalog_json = {catalog_toml} -[projects."{repo_root_display}"] +[projects.{repo_root_toml}] trust_level = "trusted" [tui.model_availability_nux] @@ -71,10 +74,16 @@ trust_level = "trusted" let fixture_path = codex_utils_cargo_bin::find_resource!("../core/tests/cli_responses_fixture.sse")?; - let codex = if let Ok(path) = codex_utils_cargo_bin::cargo_bin("codex") { + let codex = if let Some(path) = sibling_test_binary("codex") { + path + } else if let Ok(path) = codex_utils_cargo_bin::cargo_bin("codex") { path } else { - let fallback = repo_root.join("codex-rs/target/debug/codex"); + let fallback = repo_root + .join("codex-rs") + .join("target") + .join("debug") + .join(format!("codex{}", std::env::consts::EXE_SUFFIX)); if fallback.is_file() { fallback } else { @@ -195,3 +204,10 @@ trust_level = "trusted" Ok(()) } + +fn sibling_test_binary(binary: &str) -> Option { + let current_exe = std::env::current_exe().ok()?; + let profile_dir = current_exe.parent()?.parent()?; + let candidate = profile_dir.join(format!("{binary}{}", std::env::consts::EXE_SUFFIX)); + candidate.is_file().then_some(candidate) +} diff --git a/codex-rs/tui/tests/suite/model_switching_e2e.rs b/codex-rs/tui/tests/suite/model_switching_e2e.rs index cb718d3e6d6f..c5eaee61725d 100644 --- a/codex-rs/tui/tests/suite/model_switching_e2e.rs +++ b/codex-rs/tui/tests/suite/model_switching_e2e.rs @@ -839,13 +839,15 @@ async fn cross_provider_model_switch_applies_immediately_without_manual_new_sess fn tempdir_with_ollama_config(repo_root: &Path, model: &str) -> Result { let codex_home = tempfile::tempdir()?; - let repo_root_display = repo_root.display(); + let repo_root_display = repo_root.display().to_string(); + let repo_root_toml = toml::Value::String(repo_root_display).to_string(); + let model_toml = toml::Value::String(model.to_string()).to_string(); let config_contents = format!( r#"model_provider = "ollama" -model = "{model}" +model = {model_toml} cli_auth_credentials_store = "file" -[projects."{repo_root_display}"] +[projects.{repo_root_toml}] trust_level = "trusted" "# ); @@ -884,21 +886,25 @@ fn tempdir_with_github_copilot_config( serde_json::to_string(&source_catalog)?, )?; - let repo_root_display = repo_root.display(); - let catalog_display = custom_catalog_path.display(); + let repo_root_display = repo_root.display().to_string(); + let repo_root_toml = toml::Value::String(repo_root_display).to_string(); + let catalog_display = custom_catalog_path.display().to_string(); + let catalog_toml = toml::Value::String(catalog_display).to_string(); + let model_toml = toml::Value::String(model.to_string()).to_string(); + let base_url_toml = toml::Value::String(base_url.to_string()).to_string(); let config_contents = format!( r#"model_provider = "github-copilot" -model = "{model}" -model_catalog_json = "{catalog_display}" +model = {model_toml} +model_catalog_json = {catalog_toml} cli_auth_credentials_store = "file" [model_providers.github-copilot] name = "GitHub Copilot" -base_url = "{base_url}" +base_url = {base_url_toml} env_key = "GITHUB_COPILOT_TOKEN" wire_api = "responses" -[projects."{repo_root_display}"] +[projects.{repo_root_toml}] trust_level = "trusted" "# ); @@ -917,19 +923,25 @@ fn tempdir_with_models_dev_provider_config( ) -> Result { let codex_home = tempfile::tempdir()?; - let repo_root_display = repo_root.display(); + let repo_root_display = repo_root.display().to_string(); + let repo_root_toml = toml::Value::String(repo_root_display).to_string(); + let provider_id_toml = toml::Value::String(provider_id.to_string()).to_string(); + let provider_name_toml = toml::Value::String(provider_name.to_string()).to_string(); + let model_toml = toml::Value::String(model.to_string()).to_string(); + let base_url_toml = toml::Value::String(base_url.to_string()).to_string(); + let env_key_toml = toml::Value::String(env_key.to_string()).to_string(); let config_contents = format!( - r#"model_provider = "{provider_id}" -model = "{model}" + r#"model_provider = {provider_id_toml} +model = {model_toml} cli_auth_credentials_store = "file" [model_providers.{provider_id}] -name = "{provider_name}" -base_url = "{base_url}" -env_key = "{env_key}" +name = {provider_name_toml} +base_url = {base_url_toml} +env_key = {env_key_toml} wire_api = "responses" -[projects."{repo_root_display}"] +[projects.{repo_root_toml}] trust_level = "trusted" "# ); @@ -946,25 +958,29 @@ fn tempdir_with_dual_provider_config( ) -> Result { let codex_home = tempfile::tempdir()?; - let repo_root_display = repo_root.display(); + let repo_root_display = repo_root.display().to_string(); + let repo_root_toml = toml::Value::String(repo_root_display).to_string(); + let startup_model_toml = toml::Value::String(startup_model.to_string()).to_string(); + let azure_base_url_toml = toml::Value::String(azure_base_url.to_string()).to_string(); + let copilot_base_url_toml = toml::Value::String(copilot_base_url.to_string()).to_string(); let config_contents = format!( r#"model_provider = "azure-local" -model = "{startup_model}" +model = {startup_model_toml} cli_auth_credentials_store = "file" [model_providers.azure-local] name = "Azure" -base_url = "{azure_base_url}" +base_url = {azure_base_url_toml} env_key = "AZURE_TEST_KEY" wire_api = "responses" [model_providers.github-copilot] name = "GitHub Copilot" -base_url = "{copilot_base_url}" +base_url = {copilot_base_url_toml} env_key = "GITHUB_COPILOT_TOKEN" wire_api = "responses" -[projects."{repo_root_display}"] +[projects.{repo_root_toml}] trust_level = "trusted" "# ); @@ -1189,6 +1205,7 @@ async fn run_codex_cli_with_filter( .await } +#[expect(clippy::too_many_arguments)] async fn run_codex_cli_with_filter_options( codex_cli: &Path, codex_home: &Path, @@ -1364,16 +1381,27 @@ async fn run_codex_cli_with_filter_options( } fn find_codex_cli(cwd: &Path) -> Option { - // Always build a fresh local binary once per test process so PTY E2E - // assertions exercise the current source tree instead of stale artifacts. + if let Some(path) = sibling_test_binary("codex") { + return Some(path); + } + + if let Ok(path) = codex_utils_cargo_bin::cargo_bin("codex") { + return Some(path); + } + + let fallback = debug_target_binary(cwd, "codex"); + if fallback.is_file() { + return Some(fallback); + } + if ensure_fallback_codex_binary_is_built(cwd).is_ok() { - let fallback = cwd.join("codex-rs/target/debug/codex"); - if fallback.is_file() { - return Some(fallback); - } + return sibling_test_binary("codex").or_else(|| { + let fallback = debug_target_binary(cwd, "codex"); + fallback.is_file().then_some(fallback) + }); } - codex_utils_cargo_bin::cargo_bin("codex").ok() + None } fn ensure_fallback_codex_binary_is_built(repo_root: &Path) -> Result<()> { @@ -1381,6 +1409,8 @@ fn ensure_fallback_codex_binary_is_built(repo_root: &Path) -> Result<()> { let result = BUILD_RESULT.get_or_init(|| { let status = Command::new("cargo") .arg("build") + .arg("-p") + .arg("codex-cli") .arg("--bin") .arg("codex") .current_dir(repo_root.join("codex-rs")) @@ -1400,6 +1430,21 @@ fn ensure_fallback_codex_binary_is_built(repo_root: &Path) -> Result<()> { } } +fn sibling_test_binary(binary: &str) -> Option { + let current_exe = std::env::current_exe().ok()?; + let profile_dir = current_exe.parent()?.parent()?; + let candidate = profile_dir.join(format!("{binary}{}", std::env::consts::EXE_SUFFIX)); + candidate.is_file().then_some(candidate) +} + +fn debug_target_binary(repo_root: &Path, binary: &str) -> PathBuf { + repo_root + .join("codex-rs") + .join("target") + .join("debug") + .join(format!("{binary}{}", std::env::consts::EXE_SUFFIX)) +} + fn spawn_openai_compat_models_and_responses_server( models_response_json: serde_json::Value, response_model: &str, @@ -1753,13 +1798,15 @@ fn spawn_openai_compat_models_with_chat_completions_fallback_server( Ok((format!("http://{address}/v1"), handle)) } +type ResponsesServerHandle = thread::JoinHandle>>; + fn spawn_models_dev_and_responses_server( models_dev_provider_id: &str, models_dev_provider_name: &str, model_ids: &[&str], response_model: &str, answer_text: &str, -) -> Result<(String, String, thread::JoinHandle>>)> { +) -> Result<(String, String, ResponsesServerHandle)> { let listener = TcpListener::bind("127.0.0.1:0")?; let address = listener.local_addr()?; @@ -1919,11 +1966,7 @@ data: {{\"type\":\"response.completed\",\"response\":{{\"id\":\"resp-1\",\"usage Ok(requests) }); - Ok(( - provider_api.clone(), - format!("http://{address}/api.json"), - handle, - )) + Ok((provider_api, format!("http://{address}/api.json"), handle)) } async fn type_text_with_stabilization(writer: &tokio::sync::mpsc::Sender>, text: &str) { diff --git a/docs/reflection.md b/docs/reflection.md index 3afff5837b77..9b25a55008a3 100644 --- a/docs/reflection.md +++ b/docs/reflection.md @@ -7,6 +7,7 @@ The reflection layer is an experimental feature that verifies if the AI agent co 1. **Task Execution**: Agent executes the user's request using tools (shell, file operations, etc.) 2. **Evaluation**: After completion, the reflection layer: + - Collects context: original task, recent tool calls (up to 10), and final response - Sends this to a judge model for evaluation - Receives a verdict with completion status and confidence score @@ -60,6 +61,7 @@ cargo test -p codex-core --test all --release reflection_layer_hello_world -- -- ``` The test verifies: + 1. Azure OpenAI integration with reflection enabled 2. Agent creates requested Python files 3. Tests pass via pytest @@ -71,13 +73,14 @@ The eval suite measures the reflection layer's impact on coding task performance ### Tasks -| Task | Description | Bug Type | -|------|-------------|----------| +| Task | Description | Bug Type | +| ------ | ----------------- | --------------------------------------- | | Task 1 | Off-by-one errors | `range(n+1)` → `range(n)`, index errors | -| Task 2 | String logic | Palindrome detection, word counting | -| Task 3 | Edge cases | Division by zero, empty list handling | +| Task 2 | String logic | Palindrome detection, word counting | +| Task 3 | Edge cases | Division by zero, empty list handling | Each task provides: + - A buggy Python codebase - An issue description (like a GitHub issue) - Test files that verify the fix @@ -123,6 +126,7 @@ Improvement: +1 tasks ``` The reflection layer helps catch incomplete fixes by re-evaluating the agent's work and providing feedback for another attempt. + ## Local Debugging ### Prerequisites @@ -159,6 +163,7 @@ fi ``` Notes: + - Using `install -m 755` sets the executable bit and is safer than `cp`. - Avoid using `sudo` unless installing to system locations like `/usr/local/bin`. diff --git a/scripts/stage_npm_packages.py b/scripts/stage_npm_packages.py index 5bbee755e928..0e074d545465 100755 --- a/scripts/stage_npm_packages.py +++ b/scripts/stage_npm_packages.py @@ -79,13 +79,16 @@ def expand_packages(packages: list[str]) -> list[str]: def resolve_release_workflow(version: str) -> dict: + release_branch = f"rust-v{version}" stdout = subprocess.check_output( [ "gh", "run", "list", + "-R", + GITHUB_REPO, "--branch", - f"rust-v{version}", + release_branch, "--json", "workflowName,url,headSha", "--workflow", @@ -98,7 +101,10 @@ def resolve_release_workflow(version: str) -> dict: ) workflow = json.loads(stdout or "null") if not workflow: - raise RuntimeError(f"Unable to find rust-release workflow for version {version}.") + raise RuntimeError( + "Unable to find rust-release workflow for version " + f"{version} in {GITHUB_REPO} (branch {release_branch}, workflow {WORKFLOW_NAME})." + ) return workflow diff --git a/scripts/test_stage_npm_packages.py b/scripts/test_stage_npm_packages.py new file mode 100644 index 000000000000..f262f13fe315 --- /dev/null +++ b/scripts/test_stage_npm_packages.py @@ -0,0 +1,50 @@ +#!/usr/bin/env python3 + +import importlib.util +from pathlib import Path +import unittest +from unittest.mock import patch + + +SCRIPT_PATH = Path(__file__).resolve().parent / "stage_npm_packages.py" +SPEC = importlib.util.spec_from_file_location("stage_npm_packages", SCRIPT_PATH) +if SPEC is None or SPEC.loader is None: + raise RuntimeError(f"Unable to load module from {SCRIPT_PATH}") +MODULE = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(MODULE) + + +class ResolveReleaseWorkflowTests(unittest.TestCase): + def test_queries_upstream_repo_for_release_workflow(self) -> None: + expected = { + "workflowName": "rust-release", + "url": "https://github.com/openai/codex/actions/runs/20345806534", + "headSha": "5b9d9a60d74c8ee2cb34d60fb14b71990e8318ea", + } + with patch.object( + MODULE.subprocess, + "check_output", + return_value=MODULE.json.dumps(expected), + ) as check_output: + workflow = MODULE.resolve_release_workflow("0.74.0") + + self.assertEqual(workflow, expected) + cmd = check_output.call_args.args[0] + self.assertEqual(cmd[:4], ["gh", "run", "list", "-R"]) + self.assertEqual(cmd[4], MODULE.GITHUB_REPO) + self.assertIn("--branch", cmd) + self.assertEqual(cmd[cmd.index("--branch") + 1], "rust-v0.74.0") + self.assertIn("--workflow", cmd) + self.assertEqual(cmd[cmd.index("--workflow") + 1], MODULE.WORKFLOW_NAME) + + def test_missing_workflow_error_mentions_repo_and_branch(self) -> None: + with patch.object(MODULE.subprocess, "check_output", return_value=""): + with self.assertRaisesRegex( + RuntimeError, + r"openai/codex.*rust-v0\.74\.0.*\.github/workflows/rust-release\.yml", + ): + MODULE.resolve_release_workflow("0.74.0") + + +if __name__ == "__main__": + unittest.main()