diff --git a/actions/setup/js/codex_harness.cjs b/actions/setup/js/codex_harness.cjs index 9953833df30..d79cc78bf64 100644 --- a/actions/setup/js/codex_harness.cjs +++ b/actions/setup/js/codex_harness.cjs @@ -34,6 +34,9 @@ const { getErrorMessage } = require("./error_helpers.cjs"); const fs = require("fs"); +const { DEFAULT_MCP_CALL_WATCHDOG_MS, MCP_CALL_TRANSPORT_GRACE_MS } = require("./constants.cjs"); +const { loadCompiledConfig, mergeConfig } = require("./codex_config.cjs"); +const { parseJsonPrefix } = require("./parse_json_prefix.cjs"); const { runProcess, formatDuration, sleep, MIN_POST_RESULT_WATCHDOG_TIMEOUT_MS, DEFAULT_POST_RESULT_WATCHDOG_IDLE_TIMEOUT_MS, MAX_POST_RESULT_WATCHDOG_TIMEOUT_MS, resolvePostResultWatchdogIdleTimeoutMs } = require("./process_runner.cjs"); const { runHarnessRetryLoop, shouldSkipForNoopSafeOutputs, shouldStopForNoopSafeOutputs } = require("./harness_retry_runner.cjs"); const { @@ -92,6 +95,62 @@ const SERVER_ERROR_PATTERN = /InternalServerError|ServiceUnavailableError|500 In // an identical rejection: retrying only re-bills the turns that succeeded before the failure point. const INVALID_REQUEST_ERROR_PATTERN = /invalid_request_error/i; +function resolveMCPServerToolTimeouts(config, runtimeToolTimeoutSeconds) { + const configuredServers = config.defaults?.mcp_servers && typeof config.defaults.mcp_servers === "object" ? config.defaults.mcp_servers : {}; + const defaults = { + ...config.defaults, + mcp_servers: Object.fromEntries( + Object.entries(configuredServers).map(([name, value]) => [ + name, + typeof value === "object" && value !== null && !Array.isArray(value) + ? { ...value, ...(Number.isSafeInteger(runtimeToolTimeoutSeconds) && runtimeToolTimeoutSeconds > 0 ? { tool_timeout_sec: runtimeToolTimeoutSeconds } : {}) } + : value, + ]) + ), + }; + const effectiveServers = mergeConfig(defaults, config.overrides || {}).mcp_servers || {}; + return Object.fromEntries( + Object.entries(effectiveServers).flatMap(([name, value]) => (typeof value?.tool_timeout_sec === "number" && Number.isSafeInteger(value.tool_timeout_sec) && value.tool_timeout_sec > 0 ? [[name, value.tool_timeout_sec]] : [])) + ); +} + +function createMCPCallWatchdog(timeoutMs, now = Date.now) { + const pending = new Map(); + function track(eventType, item) { + if (item?.type !== "mcp_tool_call" || typeof item.id !== "string") return; + if (eventType === "item.started") { + const callTimeoutMs = typeof timeoutMs === "function" ? timeoutMs(item) : timeoutMs; + pending.set(item.id, { startedAt: now(), timeoutMs: Number.isSafeInteger(callTimeoutMs) && callTimeoutMs > 0 ? callTimeoutMs : DEFAULT_MCP_CALL_WATCHDOG_MS }); + } + if (eventType === "item.completed" || eventType === "item.failed") pending.delete(item.id); + } + return { + observe(line) { + let event; + try { + event = JSON.parse(line); + } catch { + return; + } + track(event?.type, event?.item); + }, + observePrefix(prefix) { + const event = parseJsonPrefix(prefix); + track(event?.type, event?.item); + }, + expiredTimeoutMs() { + const current = now(); + for (const call of pending.values()) { + if (current - call.startedAt >= call.timeoutMs) return call.timeoutMs; + } + return null; + }, + expired() { + return this.expiredTimeoutMs() !== null; + }, + }; +} + // Codex's `turn.failed` event nests the actual provider error as a JSON string inside // `error.message` (sometimes doubly-nested, e.g. `error.message` -> `{"error": {...}}`). // This is a specific, common form of "unsupported model" failure: the configured model does @@ -785,6 +844,9 @@ async function main() { // The deadline includes preflight time and is checked both between and during attempts. const softTimeoutGuard = buildSoftTimeoutGuard(driverStartTime); const contextRebuildCircuitBreaker = resolveContextRebuildCircuitBreakerConfig(process.env); + const configuredToolTimeout = Number(codexEnv.GH_AW_TOOL_TIMEOUT); + const fallbackToolTimeoutMs = Number.isSafeInteger(configuredToolTimeout) && configuredToolTimeout > 0 ? configuredToolTimeout * 1000 + MCP_CALL_TRANSPORT_GRACE_MS : DEFAULT_MCP_CALL_WATCHDOG_MS; + const serverToolTimeouts = resolveMCPServerToolTimeouts(loadCompiledConfig(), configuredToolTimeout); /** @type {string[] | null} */ let resumeArgs = null; let lastThreadId = ""; @@ -808,6 +870,10 @@ async function main() { getRetryMode: () => (resumeArgs ? `resume ${lastThreadId}` : "fresh run"), runAttempt: async attempt => { const terminalErrors = []; + const mcpWatchdog = createMCPCallWatchdog(item => { + const server = item.server ?? item.server_name ?? item.serverName; + return serverToolTimeouts[server] ? serverToolTimeouts[server] * 1000 + MCP_CALL_TRANSPORT_GRACE_MS : fallbackToolTimeoutMs; + }); let nextContextCheckAt = Date.now() + contextRebuildCircuitBreaker.pollIntervalMs; // Track the file size before this attempt so the watchdog only arms on output // written by this attempt, not by a previous retry. @@ -823,6 +889,7 @@ async function main() { stdin: resumeArgs ? resumePrompt : promptInput.stdin, maxCollectedOutputBytes: 4 * 1024 * 1024, onStdoutLine: line => { + mcpWatchdog.observe(line); try { const event = JSON.parse(line); if (event.type === "thread.started" && typeof event.thread_id === "string") lastThreadId = event.thread_id; @@ -832,27 +899,36 @@ async function main() { } } catch {} }, - runtimeGuard: - contextRebuildCircuitBreaker.enabled || softTimeoutGuard - ? { - pollIntervalMs: Math.min(contextRebuildCircuitBreaker.pollIntervalMs, 1000), - termGraceMs: contextRebuildCircuitBreaker.termGraceMs, - shouldTerminate: async () => { - if (softTimeoutGuard && Date.now() >= softTimeoutGuard.softDeadlineMs) return { terminate: true, reason: "Codex reached the soft execution deadline; stopping to preserve structured output before the step timeout." }; - if (!contextRebuildCircuitBreaker.enabled) return false; - if (Date.now() < nextContextCheckAt) return false; - nextContextCheckAt = Date.now() + contextRebuildCircuitBreaker.pollIntervalMs; - return evaluateContextRebuildCircuitBreakerForAttempt( - await readWorkingSetFromTokenUsage(tokenUsagePaths), - { - maxRebuildFactor: contextRebuildCircuitBreaker.maxRebuildFactor, - minCumulativeInputTokens: contextRebuildCircuitBreaker.minCumulativeInputTokens, - }, - { safeOutputsPath, safeOutputsByteOffset, logger: log } - ); - }, - } - : undefined, + onStdoutLinePrefix: prefix => mcpWatchdog.observePrefix(prefix), + runtimeGuard: { + pollIntervalMs: Math.min(contextRebuildCircuitBreaker.pollIntervalMs, 1000), + termGraceMs: contextRebuildCircuitBreaker.termGraceMs, + onTriggered: decision => { + if (decision.event) process.stdout.write(`${JSON.stringify(decision.event)}\n`); + }, + shouldTerminate: async () => { + if (softTimeoutGuard && Date.now() >= softTimeoutGuard.softDeadlineMs) return { terminate: true, reason: "Codex reached the soft execution deadline; stopping to preserve structured output before the step timeout." }; + const expiredMCPCallTimeoutMs = mcpWatchdog.expiredTimeoutMs(); + if (expiredMCPCallTimeoutMs !== null) { + return { + terminate: true, + reason: `transport_wedge: MCP tool call timed out after ${Math.round(expiredMCPCallTimeoutMs / 1000)}s`, + event: { type: "agent.execution", data: { categories: ["transport_wedge"], errorCodes: [], errorTypes: [] } }, + }; + } + if (!contextRebuildCircuitBreaker.enabled) return false; + if (Date.now() < nextContextCheckAt) return false; + nextContextCheckAt = Date.now() + contextRebuildCircuitBreaker.pollIntervalMs; + return evaluateContextRebuildCircuitBreakerForAttempt( + await readWorkingSetFromTokenUsage(tokenUsagePaths), + { + maxRebuildFactor: contextRebuildCircuitBreaker.maxRebuildFactor, + minCumulativeInputTokens: contextRebuildCircuitBreaker.minCumulativeInputTokens, + }, + { safeOutputsPath, safeOutputsByteOffset, logger: log } + ); + }, + }, postResultWatchdog: safeOutputsPath ? { shouldArm: () => @@ -1108,6 +1184,8 @@ if (typeof module !== "undefined" && module.exports) { injectModelFlagAfterExec, getCodexModelEnvVar, resolvePostResultWatchdogIdleTimeoutMs, + createMCPCallWatchdog, + resolveMCPServerToolTimeouts, POST_RESULT_WATCHDOG_IDLE_TIMEOUT_MS, DEFAULT_POST_RESULT_WATCHDOG_IDLE_TIMEOUT_MS, MIN_POST_RESULT_WATCHDOG_TIMEOUT_MS, diff --git a/actions/setup/js/codex_harness.test.cjs b/actions/setup/js/codex_harness.test.cjs index b1e25ce4b4f..e217e0ac0b1 100644 --- a/actions/setup/js/codex_harness.test.cjs +++ b/actions/setup/js/codex_harness.test.cjs @@ -41,6 +41,8 @@ const { DEFAULT_CONTEXT_REBUILD_POLL_INTERVAL_MS, DEFAULT_CONTEXT_REBUILD_TERM_GRACE_MS, resolvePostResultWatchdogIdleTimeoutMs, + createMCPCallWatchdog, + resolveMCPServerToolTimeouts, DEFAULT_POST_RESULT_WATCHDOG_IDLE_TIMEOUT_MS, MIN_POST_RESULT_WATCHDOG_TIMEOUT_MS, MAX_POST_RESULT_WATCHDOG_TIMEOUT_MS, @@ -117,6 +119,61 @@ function runHarnessFixture(script, { prompt = "fix the bug", args = [], env = {} } describe("codex_harness.cjs", () => { + describe("MCP call watchdog", () => { + it("times out only an outstanding MCP call, not other output or completed calls", () => { + let time = 0; + const watchdog = createMCPCallWatchdog(120_000, () => time); + watchdog.observe(JSON.stringify({ type: "item.started", item: { type: "mcp_tool_call", id: "1", name: "search_repositories" } })); + time = 119_999; + watchdog.observe(JSON.stringify({ type: "item.completed", item: { type: "agent_message", id: "other" } })); + expect(watchdog.expired()).toBe(false); + time = 120_000; + expect(watchdog.expired()).toBe(true); + watchdog.observe(JSON.stringify({ type: "item.completed", item: { type: "mcp_tool_call", id: "1" } })); + expect(watchdog.expired()).toBe(false); + }); + + it("clears failed calls and ignores malformed events", () => { + let time = 0; + const watchdog = createMCPCallWatchdog(100, () => time); + watchdog.observe("{"); + watchdog.observe(JSON.stringify({ type: "item.started", item: { type: "mcp_tool_call", id: "2" } })); + watchdog.observe(JSON.stringify({ type: "item.failed", item: { type: "mcp_tool_call", id: "2" } })); + time = 200; + expect(watchdog.expired()).toBe(false); + }); + + it("uses per-server timeouts after the global timeout override", () => { + const timeouts = resolveMCPServerToolTimeouts( + { + defaults: { mcp_servers: { github: { tool_timeout_sec: 60 }, search: { tool_timeout_sec: 30 } } }, + overrides: { mcp_servers: { github: { tool_timeout_sec: 180 } } }, + }, + 90 + ); + expect(timeouts).toEqual({ github: 180, search: 90 }); + + let time = 0; + const watchdog = createMCPCallWatchdog( + item => timeouts[item.server] * 1000 + 60_000, + () => time + ); + watchdog.observe(JSON.stringify({ type: "item.started", item: { type: "mcp_tool_call", id: "1", server: "github" } })); + time = 239_999; + expect(watchdog.expired()).toBe(false); + time = 240_000; + expect(watchdog.expiredTimeoutMs()).toBe(240_000); + }); + + it("clears an oversized completion from its bounded lifecycle prefix", () => { + let time = 0; + const watchdog = createMCPCallWatchdog(100, () => time); + watchdog.observe(JSON.stringify({ type: "item.started", item: { type: "mcp_tool_call", id: "large" } })); + time = 100; + watchdog.observePrefix('{"type":"item.com\\u0070leted","item":{"\\u0069d":"large","type":"mcp_tool_call","server":"github","result":"'); + expect(watchdog.expired()).toBe(false); + }); + }); describe("native exec orchestration", () => { it("preserves a native argument-parse exit without retrying a deterministic startup error", () => { const { result, calls } = runHarnessFixture(`process.stderr.write("error: unexpected argument '--invalid' found\\n\\nUsage: codex exec [OPTIONS] [PROMPT]\\n");process.exit(2);`); diff --git a/actions/setup/js/constants.cjs b/actions/setup/js/constants.cjs index c136c9d8f88..7a4e20ba027 100644 --- a/actions/setup/js/constants.cjs +++ b/actions/setup/js/constants.cjs @@ -156,6 +156,18 @@ const DETECTION_LOG_FILENAME = "detection.log"; */ const DETECTION_RESULT_FILENAME = "detection_result.json"; +/** + * Default timeout for a Codex MCP tool call when no server timeout is configured. + * @type {number} + */ +const DEFAULT_MCP_CALL_WATCHDOG_MS = 120_000; + +/** + * Grace period added to the configured MCP tool timeout before terminating Codex. + * @type {number} + */ +const MCP_CALL_TRANSPORT_GRACE_MS = 60_000; + module.exports = { AGENT_OUTPUT_FILENAME, TMP_GH_AW_PATH, @@ -175,4 +187,6 @@ module.exports = { GITHUB_RATE_LIMITS_JSONL_PATH, DETECTION_LOG_FILENAME, DETECTION_RESULT_FILENAME, + DEFAULT_MCP_CALL_WATCHDOG_MS, + MCP_CALL_TRANSPORT_GRACE_MS, }; diff --git a/actions/setup/js/constants.test.cjs b/actions/setup/js/constants.test.cjs index 1bfd07f6e1f..f88a4e3e416 100644 --- a/actions/setup/js/constants.test.cjs +++ b/actions/setup/js/constants.test.cjs @@ -13,6 +13,8 @@ const { MANIFEST_FILE_PATH, TEMPORARY_ID_MAP_FILE_PATH, DETECTION_LOG_FILENAME, + DEFAULT_MCP_CALL_WATCHDOG_MS, + MCP_CALL_TRANSPORT_GRACE_MS, } = require("./constants.cjs"); describe("constants", () => { @@ -83,6 +85,13 @@ describe("constants", () => { }); }); + describe("Codex MCP watchdog timeouts", () => { + it("should export the default call timeout and transport grace period", () => { + expect(DEFAULT_MCP_CALL_WATCHDOG_MS).toBe(120_000); + expect(MCP_CALL_TRANSPORT_GRACE_MS).toBe(60_000); + }); + }); + describe("module exports", () => { it("should export all expected constants", () => { const exported = require("./constants.cjs"); @@ -99,6 +108,8 @@ describe("constants", () => { "MANIFEST_FILE_PATH", "TEMPORARY_ID_MAP_FILE_PATH", "DETECTION_LOG_FILENAME", + "DEFAULT_MCP_CALL_WATCHDOG_MS", + "MCP_CALL_TRANSPORT_GRACE_MS", ]; for (const key of expectedKeys) { expect(exported).toHaveProperty(key); diff --git a/actions/setup/js/handle_agent_failure.cjs b/actions/setup/js/handle_agent_failure.cjs index 31ece7aa515..a229f106b87 100644 --- a/actions/setup/js/handle_agent_failure.cjs +++ b/actions/setup/js/handle_agent_failure.cjs @@ -31,6 +31,7 @@ const { extractShellCommandFromToolData } = require("./tool_call_details.cjs"); const { resolveFailureIssueRepo } = require("./repo_helpers.cjs"); const { GITHUB_API_VERSION } = require("./constants.cjs"); const { EMPTY_OUTPUT_CAUSES } = require("./empty_output_outcome.cjs"); +const { isAgentExecutionEvent } = require("./agent_execution.cjs"); const fs = require("fs"); const https = require("https"); const os = require("os"); @@ -285,6 +286,7 @@ function parseHTMLCommentMetadata(body, markerKey) { function buildFailureMatchCategories(options) { const categories = []; + if (options.transportWedge) categories.push("transport_wedge"); if (options.isTimedOut) categories.push("timed_out"); if (options.hasAssignmentErrors) categories.push("assignment_errors"); if (options.hasAssignCopilotFailures) categories.push("assign_copilot_failures"); @@ -331,11 +333,32 @@ function buildFailureMatchCategories(options) { return categories.sort(); } +function hasMCPTransportWedge(sessionContent) { + if (!sessionContent.includes('"transport_wedge"')) return false; + return sessionContent.split(/\r?\n/).some(line => { + try { + const event = JSON.parse(line); + return isAgentExecutionEvent(event) && event.data.categories.includes("transport_wedge"); + } catch { + return false; + } + }); +} + +function getAgentStdioLogPath(agentOutputFile = process.env.GH_AW_AGENT_OUTPUT) { + return agentOutputFile ? path.join(path.dirname(agentOutputFile), "agent-stdio.log") : "/tmp/gh-aw/agent-stdio.log"; +} + +function getAgentSessionPath(agentOutputFile = process.env.GH_AW_AGENT_OUTPUT) { + return agentOutputFile ? path.join(path.dirname(agentOutputFile), "agent-session.jsonl") : "/tmp/gh-aw/agent-session.jsonl"; +} + /** * Build a precise failure issue title for known failure classes. * Falls back to the generic failure title when no specific class matches. * @param {Object} options * @param {string} options.workflowName + * @param {boolean} [options.transportWedge] * @param {boolean} options.isTimedOut * @param {boolean} options.hasMissingSafeOutputs * @param {boolean} options.hasReportIncomplete @@ -392,6 +415,7 @@ function buildFailureIssueTitle(options) { const agentName = sanitizeContent(options.copilotAgentNotFound, COPILOT_AGENT_NOT_FOUND_AGENT_MAX_LENGTH).replace(/\s+/g, " ").trim(); return `[aw] ${workflowName} could not find configured Copilot agent "${agentName}"`; } + if (options.transportWedge) return `[aw] ${workflowName} stalled on an MCP tool call`; if (options.isTimedOut) return `[aw] ${workflowName} timed out`; if (options.hasToolDenialsExceeded) return `[aw] ${workflowName} exceeded tool denial limit`; if (options.hasCacheMissMisconfiguration) return `[aw] ${workflowName} has cache-memory miss misconfiguration`; @@ -4276,10 +4300,19 @@ async function main() { // Sanitize workflow name for title const sanitizedWorkflowName = sanitizeContent(workflowName, { maxLength: 100 }); + let transportWedge = false; + if (agentConclusion === "failure") { + try { + transportWedge = hasMCPTransportWedge(fs.readFileSync(getAgentSessionPath(), "utf8")); + } catch { + core.debug("Unified agent session unavailable for MCP watchdog classification"); + } + } // Only the collector-written root metadata is trusted; report_incomplete.reason is agent-controlled. const emptyOutputCause = agentOutputResult.success ? agentOutputResult.collectorEmptyOutputCause : undefined; const issueTitle = buildFailureIssueTitle({ workflowName: sanitizedWorkflowName, + transportWedge, emptyOutputCause, isTimedOut, hasMissingSafeOutputs, @@ -4308,6 +4341,7 @@ async function main() { }); const failureCategories = buildFailureMatchCategories({ agentConclusion, + transportWedge, emptyOutputCause, isTimedOut, hasAssignmentErrors, @@ -5019,6 +5053,9 @@ module.exports = { CASCADE_ROLLUP_TITLE, FAILURE_TITLE_PATTERN, buildFailureMatchCategories, + hasMCPTransportWedge, + getAgentStdioLogPath, + getAgentSessionPath, buildFailureIssueTitle, FAILURE_CATEGORIES_PATH, }; diff --git a/actions/setup/js/handle_agent_failure.test.cjs b/actions/setup/js/handle_agent_failure.test.cjs index 2b656715cdb..372309e6dd6 100644 --- a/actions/setup/js/handle_agent_failure.test.cjs +++ b/actions/setup/js/handle_agent_failure.test.cjs @@ -420,6 +420,7 @@ describe("handle_agent_failure", () => { it("falls back to generic failed title when no specific condition matches", () => { expect(buildFailureIssueTitle(baseOptions)).toBe("[aw] Test Workflow failed"); + expect(buildFailureIssueTitle({ ...baseOptions, transportWedge: true })).toBe("[aw] Test Workflow stalled on an MCP tool call"); }); it.each([ @@ -6408,6 +6409,24 @@ describe("handle_agent_failure", () => { expect(categories).toContain("timed_out"); }); + it("classifies an MCP watchdog failure without a generic agent failure", () => { + expect(buildFailureMatchCategories({ agentConclusion: "failure", transportWedge: true })).toEqual(["transport_wedge"]); + }); + + it("classifies MCP watchdog failures from structured unified-session events", () => { + const { hasMCPTransportWedge, getAgentStdioLogPath, getAgentSessionPath } = require("./handle_agent_failure.cjs"); + const event = { type: "agent.execution", data: { categories: ["transport_wedge"], errorCodes: [], errorTypes: [] } }; + expect(hasMCPTransportWedge(JSON.stringify(event))).toBe(true); + expect(hasMCPTransportWedge(JSON.stringify({ type: "agent.execution", data: { categories: ["agent_failure"], errorCodes: [], errorTypes: [] } }))).toBe(false); + expect(hasMCPTransportWedge("not-json\n".repeat(10_000))).toBe(false); + expect(hasMCPTransportWedge('{"type":"item.failed","message":"transport_wedge: MCP tool call timed out after 120s"}')).toBe(false); + expect(hasMCPTransportWedge("[codex-harness] runtime guard requested termination (transport_wedge: MCP tool call timed out after 120s)")).toBe(false); + expect(getAgentStdioLogPath("/tmp/call-workflow/agent_output.json")).toBe("/tmp/call-workflow/agent-stdio.log"); + expect(getAgentStdioLogPath("")).toBe("/tmp/gh-aw/agent-stdio.log"); + expect(getAgentSessionPath("/tmp/call-workflow/agent_output.json")).toBe("/tmp/call-workflow/agent-session.jsonl"); + expect(getAgentSessionPath("")).toBe("/tmp/gh-aw/agent-session.jsonl"); + }); + it("returns missing_safe_outputs category", () => { const categories = buildFailureMatchCategories({ hasMissingSafeOutputs: true, diff --git a/actions/setup/js/parse_codex_log.test.cjs b/actions/setup/js/parse_codex_log.test.cjs index 9bb133e1144..bfdbc6466b3 100644 --- a/actions/setup/js/parse_codex_log.test.cjs +++ b/actions/setup/js/parse_codex_log.test.cjs @@ -813,6 +813,11 @@ ERROR: This user's access to o4-mini has been temporarily limited`; expect(isCodexJsonlFormat("thinking\nsome thinking\ntool github.x({})".split("\n"))).toBe(false); }); + it("preserves structured runtime-guard execution events", () => { + const event = { type: "agent.execution", data: { categories: ["transport_wedge"], errorCodes: [], errorTypes: [] } }; + expect(parseCodexLog(JSON.stringify(event)).logEntries).toContainEqual(event); + }); + it.each([ { type: "reasoning", data: { content: "legacy reasoning" } }, { type: "assistant", message: { content: [{ type: "text", text: "legacy response" }] } }, diff --git a/actions/setup/js/parse_json_prefix.cjs b/actions/setup/js/parse_json_prefix.cjs new file mode 100644 index 00000000000..9d82028e6a0 --- /dev/null +++ b/actions/setup/js/parse_json_prefix.cjs @@ -0,0 +1,105 @@ +// @ts-check + +/** + * Parses the JSON value at the start of a possibly truncated input string. + * Completed object properties are retained when a nested value is incomplete. + * @param {string} input + * @returns {any} + */ +function parseJsonPrefix(input) { + const incomplete = Symbol("incomplete"); + let index = 0; + + function isWhitespace(character) { + return character === " " || character === "\t" || character === "\n" || character === "\r"; + } + + function skipWhitespace() { + while (isWhitespace(input[index])) index++; + } + + function parseString() { + const start = index++; + let escaped = false; + while (index < input.length) { + const character = input[index++]; + if (escaped) { + escaped = false; + } else if (character === "\\") { + escaped = true; + } else if (character === '"') { + try { + return JSON.parse(input.slice(start, index)); + } catch { + return incomplete; + } + } + } + return incomplete; + } + + function parseValue(depth = 0) { + if (depth > 64) throw new Error("JSON nesting limit exceeded"); + skipWhitespace(); + const character = input[index]; + if (character === '"') { + const value = parseString(); + return { value, complete: value !== incomplete }; + } + if (character === "{" || character === "[") { + const isObject = character === "{"; + const endCharacter = isObject ? "}" : "]"; + const value = isObject ? Object.create(null) : []; + index++; + skipWhitespace(); + if (input[index] === endCharacter) { + index++; + return { value, complete: true }; + } + while (index < input.length) { + let key; + if (isObject) { + if (input[index] !== '"') return { value, complete: false }; + key = parseString(); + if (key === incomplete) return { value, complete: false }; + skipWhitespace(); + if (input[index++] !== ":") return { value, complete: false }; + } + const child = parseValue(depth + 1); + if (child.value !== incomplete) { + if (isObject) value[key] = child.value; + else value.push(child.value); + } + if (!child.complete) return { value, complete: false }; + skipWhitespace(); + if (input[index] === endCharacter) { + index++; + return { value, complete: true }; + } + if (input[index++] !== ",") return { value, complete: false }; + skipWhitespace(); + } + return { value, complete: false }; + } + + const start = index; + while (index < input.length && !isWhitespace(input[index]) && !",[]{}".includes(input[index])) { + index++; + } + if (start === index) return { value: incomplete, complete: false }; + try { + return { value: JSON.parse(input.slice(start, index)), complete: true }; + } catch { + return { value: incomplete, complete: false }; + } + } + + try { + const result = parseValue().value; + return result === incomplete ? undefined : result; + } catch { + return undefined; + } +} + +module.exports = { parseJsonPrefix }; diff --git a/actions/setup/js/parse_json_prefix.test.cjs b/actions/setup/js/parse_json_prefix.test.cjs new file mode 100644 index 00000000000..d37b9d22613 --- /dev/null +++ b/actions/setup/js/parse_json_prefix.test.cjs @@ -0,0 +1,93 @@ +import { describe, expect, it } from "vitest"; +import { createRequire } from "module"; + +const require = createRequire(import.meta.url); +const { parseJsonPrefix } = require("./parse_json_prefix.cjs"); + +describe("parseJsonPrefix", () => { + it.each([ + ["object", '{"type":"item.started","item":{"id":"1"}}', { type: "item.started", item: { id: "1" } }], + ["array", '[1,true,null,"text"]', [1, true, null, "text"]], + ["string", '"escaped \\"text\\""', 'escaped "text"'], + ["number", "-1.25e2", -125], + ["boolean", "false", false], + ["null", "null", null], + ])("parses a complete JSON %s", (_name, input, expected) => { + expect(parseJsonPrefix(input)).toEqual(expected); + }); + + it("decodes escaped object keys and string values", () => { + expect(parseJsonPrefix('{"ty\\u0070e":"item.com\\u0070leted","value":"\\u2603"}')).toEqual({ + type: "item.completed", + value: "☃", + }); + }); + + it("retains completed properties when a nested result is truncated", () => { + const result = parseJsonPrefix('{"type":"item.completed","item":{"id":"large","type":"mcp_tool_call","result":"'); + expect(result).toEqual({ + type: "item.completed", + item: { id: "large", type: "mcp_tool_call" }, + }); + }); + + it("retains completed array elements before a truncated element", () => { + expect(parseJsonPrefix('{"items":[1,{"complete":true,"text":"unfinished')).toEqual({ + items: [1, { complete: true }], + }); + }); + + it("keeps completed properties when the following key or value is malformed", () => { + expect(parseJsonPrefix('{"type":"item.started","item":')).toEqual({ type: "item.started" }); + expect(parseJsonPrefix('{"type":"item.started",?')).toEqual({ type: "item.started" }); + expect(parseJsonPrefix('{"type":"item.started","broken"')).toEqual({ type: "item.started" }); + }); + + it("does not expose partially parsed strings or invalid scalar values", () => { + expect(parseJsonPrefix('{"type":"item.started","id":"unfinished')).toEqual({ type: "item.started" }); + expect(parseJsonPrefix('{"type":"item.started","value":truX')).toEqual({ type: "item.started" }); + }); + + it("returns undefined for empty, whitespace-only, and invalid input", () => { + expect(parseJsonPrefix("")).toBeUndefined(); + expect(parseJsonPrefix(" \t\r\n")).toBeUndefined(); + expect(parseJsonPrefix("not-json")).toBeUndefined(); + expect(parseJsonPrefix("{invalid")).toEqual({}); + }); + + it("returns already parsed properties for mismatched delimiters", () => { + expect(parseJsonPrefix('{"type":"item.started","item":[1,2}')).toEqual({ + type: "item.started", + item: [1, 2], + }); + }); + + it("parses the first JSON value and ignores any following content", () => { + expect(parseJsonPrefix('{"type":"item.started"} trailing payload')).toEqual({ type: "item.started" }); + }); + + it("preserves __proto__ as data without modifying object prototypes", () => { + const result = parseJsonPrefix('{"__proto__":{"polluted":true},"type":"item.started"}'); + expect(Object.hasOwn(result, "__proto__")).toBe(true); + expect(result.__proto__).toEqual({ polluted: true }); + expect({}.polluted).toBeUndefined(); + }); + + it("supports nested JSON up to the configured depth limit", () => { + const input = `${"[".repeat(64)}0${"]".repeat(64)}`; + expect(parseJsonPrefix(input)).toEqual(JSON.parse(input)); + }); + + it("rejects excessive nesting without throwing", () => { + expect(parseJsonPrefix(`${"[".repeat(65)}0${"]".repeat(65)}`)).toBeUndefined(); + }); + + it("handles a large truncated payload with bounded parser input", () => { + const prefix = '{"type":"item.completed","item":{"id":"large","type":"mcp_tool_call","result":"'; + const result = parseJsonPrefix(prefix + "x".repeat(64 * 1024)); + expect(result).toEqual({ + type: "item.completed", + item: { id: "large", type: "mcp_tool_call" }, + }); + }); +}); diff --git a/actions/setup/js/process_runner.cjs b/actions/setup/js/process_runner.cjs index 5de108caf2a..ae353a3b175 100644 --- a/actions/setup/js/process_runner.cjs +++ b/actions/setup/js/process_runner.cjs @@ -79,6 +79,7 @@ function sleep(ms) { * env?: NodeJS.ProcessEnv, * stdin?: string | Buffer, * onStdoutLine?: (line: string) => void, + * onStdoutLinePrefix?: (prefix: string) => void, * onStderrLine?: (line: string) => void, * shutdownGraceMs?: number, * drainTimeoutMs?: number, @@ -90,9 +91,10 @@ function sleep(ms) { * termGraceMs?: number * }, * runtimeGuard?: { - * shouldTerminate: () => boolean | { terminate: boolean, reason?: string } | Promise, + * shouldTerminate: () => boolean | { terminate: boolean, reason?: string, event?: Record } | Promise }>, * pollIntervalMs?: number, - * termGraceMs?: number + * termGraceMs?: number, + * onTriggered?: (decision: { terminate: true, reason?: string, event?: Record }) => void * }, * stallWarningIntervalMs?: number * }} options @@ -118,6 +120,7 @@ function runProcess({ env, stdin, onStdoutLine, + onStdoutLinePrefix, onStderrLine, shutdownGraceMs = 5000, drainTimeoutMs = 2000, @@ -214,7 +217,7 @@ function runProcess({ // Complete output still goes to the parent's log streams. Keep a bounded classifier tail. return Buffer.from(combined).subarray(-outputLimit).toString("utf8"); }; - const lineObserver = callback => { + const lineObserver = (callback, onOversizedLine) => { let pending = ""; let dropping = false; return (text, final = false) => { @@ -222,6 +225,7 @@ function runProcess({ for (const part of text.split(/(?<=\n)/)) { if (!dropping) pending += part; if (pending.length > 1024 * 1024) { + onOversizedLine?.(pending.slice(0, 64 * 1024)); pending = ""; dropping = true; } @@ -234,7 +238,7 @@ function runProcess({ if (final) pending = ""; }; }; - const observeStdout = lineObserver(onStdoutLine); + const observeStdout = lineObserver(onStdoutLine, onStdoutLinePrefix); const observeStderr = lineObserver(onStderrLine); /** @param {NodeJS.Signals} signal */ function signalTree(signal) { @@ -473,6 +477,13 @@ function runProcess({ if (!terminate) return; runtimeGuardFired = true; runtimeGuardReason = typeof decision === "object" && decision !== null && typeof decision.reason === "string" ? decision.reason : ""; + if (typeof decision === "object" && decision !== null) { + try { + runtimeGuard.onTriggered?.({ ...decision, terminate: true }); + } catch { + log(`attempt ${attempt + 1}: runtime guard trigger callback failed`); + } + } const reasonSuffix = runtimeGuardReason ? ` (${runtimeGuardReason})` : ""; guardSentSigtermAt = Date.now(); log(`attempt ${attempt + 1}: runtime guard requested termination${reasonSuffix} (SIGTERM)`); diff --git a/actions/setup/js/process_runner.test.cjs b/actions/setup/js/process_runner.test.cjs index a1f69487229..d19cdc7d32a 100644 --- a/actions/setup/js/process_runner.test.cjs +++ b/actions/setup/js/process_runner.test.cjs @@ -139,6 +139,32 @@ describe("process_runner.cjs", () => { } }); + it("preserves a bounded stdout prefix when dropping an oversized line", async () => { + const observedLines = []; + const observedPrefixes = []; + const original = process.stdout.write; + process.stdout.write = () => true; + try { + const result = await runProcess({ + command: process.execPath, + args: ["-e", `process.stdout.write(JSON.stringify({type:"item.completed",item:{id:"call-1",type:"mcp_tool_call",server:"github",result:"x".repeat(1048576)}})+"\\n")`], + attempt: 0, + log: () => {}, + maxCollectedOutputBytes: 1024, + onStdoutLine: line => observedLines.push(line), + onStdoutLinePrefix: prefix => observedPrefixes.push(prefix), + }); + expect(result.exitCode).toBe(0); + expect(observedLines).toEqual([]); + expect(observedPrefixes).toHaveLength(1); + expect(observedPrefixes[0]).toContain('"id":"call-1"'); + expect(observedPrefixes[0]).toContain('"server":"github"'); + expect(observedPrefixes[0]).toHaveLength(64 * 1024); + } finally { + process.stdout.write = original; + } + }); + it("kills descendants holding inherited stdio when the runtime guard fires", async () => { const result = await runProcess({ command: process.execPath, @@ -485,6 +511,7 @@ describe("process_runner.cjs", () => { it("terminates a running process when runtime guard requests it", async () => { const logs = []; + const triggered = []; let checks = 0; const result = await runProcess({ command: process.execPath, @@ -495,8 +522,9 @@ describe("process_runner.cjs", () => { shouldTerminate: () => { checks += 1; if (checks < 3) return false; - return { terminate: true, reason: "test guard tripped" }; + return { terminate: true, reason: "test guard tripped", event: { type: "agent.execution", data: { categories: ["test_guard"], errorCodes: [], errorTypes: [] } } }; }, + onTriggered: decision => triggered.push(decision), pollIntervalMs: 25, termGraceMs: 200, }, @@ -505,6 +533,7 @@ describe("process_runner.cjs", () => { expect(result.runtimeGuardFired).toBe(true); expect(result.runtimeGuardReason).toContain("test guard tripped"); expect(result.watchdogFired).toBe(false); + expect(triggered).toEqual([{ terminate: true, reason: "test guard tripped", event: { type: "agent.execution", data: { categories: ["test_guard"], errorCodes: [], errorTypes: [] } } }]); expect(logs.some(line => line.includes("runtime guard requested termination"))).toBe(true); }); diff --git a/actions/setup/js/unified_session.test.cjs b/actions/setup/js/unified_session.test.cjs index 125f58af9c0..bcef6e9bf3a 100644 --- a/actions/setup/js/unified_session.test.cjs +++ b/actions/setup/js/unified_session.test.cjs @@ -694,6 +694,13 @@ describe("Unified conclusion session", () => { expect(published.some(event => event.type === "session.collection_warning")).toBe(false); }); + it("retains a structured Codex transport-wedge execution event in the unified session", () => { + const execution = { type: "agent.execution", data: { categories: ["transport_wedge"], errorCodes: [], errorTypes: [] } }; + write("agent-stdio.log", [JSON.stringify({ type: "thread.started", thread_id: "thread" }), JSON.stringify(execution)].join("\n")); + const events = writeUnifiedSession({ rootDir: root, engine: "codex" }); + expect(events.filter(event => event.type === "agent.execution")).toMatchObject([{ data: execution.data, provenance: { component: "execution", phase: "agent", path: "agent-stdio.log" } }]); + }); + it.each(["", "{bad\n", '{"type":"result","usage":{}}\n'])("falls back from an unusable canonical session (%j) to native events", content => { const nativePath = "sandbox/agent/logs/copilot-session-state/uuid/events.jsonl"; write("agent-session.jsonl", content); diff --git a/docs/src/content/docs/reference/frontmatter-full.md b/docs/src/content/docs/reference/frontmatter-full.md index 82f9f13372f..abd33e732ef 100644 --- a/docs/src/content/docs/reference/frontmatter-full.md +++ b/docs/src/content/docs/reference/frontmatter-full.md @@ -4347,6 +4347,16 @@ tools: # GitHub Issues; this requires the GH_AW_WORK_QUEUE_HMAC_SECRET repository secret. storage: "git" + # Fail closed and block safe outputs when no trusted inbound worker assignment is + # present. + # (optional) + require-assignment: true + + # Declare this workflow as a work-queue worker eligible to receive trusted queue + # claims. + # (optional) + worker: true + # Cache memory MCP configuration for persistent memory storage # (optional) # Accepted formats: @@ -8522,9 +8532,12 @@ safe-outputs: # Protected files prevent workflow run approval when modified by the associated # pull request. Use exclude to remove filenames or path prefixes from the default - # protected set. + # protected set, or a leading / for an exact repository path. # (optional) protected-files: + # Basenames (e.g. AGENTS.md) and path prefixes (e.g. .agents/) remove protection + # everywhere they match. A leading / excludes only the exact repository path (e.g. + # /pyproject.toml), leaving same-named nested files protected. # (optional) exclude: [] # Array of strings @@ -9918,8 +9931,8 @@ safe-outputs: # List of filenames or path prefixes to remove from the default protected-file # set. Items are matched by basename (e.g. "AGENTS.md") or path prefix (e.g. - # ".agents/"). Use this to allow the agent to modify specific files that are - # otherwise blocked by default. + # ".agents/"). A leading / matches only the exact repository path (e.g. + # "/pyproject.toml"), keeping nested files with the same name protected. # (optional) exclude: [] # Array of strings @@ -17013,8 +17026,8 @@ safe-outputs: # List of filenames or path prefixes to remove from the default protected-file # set. Items are matched by basename (e.g. "AGENTS.md") or path prefix (e.g. - # ".agents/"). Use this to allow the agent to modify specific files that are - # otherwise blocked by default. + # ".agents/"). A leading / matches only the exact repository path (e.g. + # "/pyproject.toml"), keeping nested files with the same name protected. # (optional) exclude: [] # Array of strings @@ -22226,9 +22239,10 @@ safe-outputs: # those categories trigger issues. If only prefixed (excluded) categories are # specified, all categories except those trigger issues. If both are specified, # categories must match included AND not match excluded. Common categories: - # agent_failure, timed_out, missing_safe_outputs, report_incomplete, missing_tool, - # missing_data, inference_access_error, mcp_policy_error, - # ai_credits_rate_limit_error, max_ai_credits_exceeded, daily_ai_credits_unknown. + # agent_failure, timed_out, transport_wedge, missing_safe_outputs, + # report_incomplete, missing_tool, missing_data, inference_access_error, + # mcp_policy_error, ai_credits_rate_limit_error, max_ai_credits_exceeded, + # daily_ai_credits_unknown. report-failure-as-issue: [] # Array items: string diff --git a/pkg/parser/schemas/main_workflow_schema.json b/pkg/parser/schemas/main_workflow_schema.json index 692488f178a..7ffa91cb82d 100644 --- a/pkg/parser/schemas/main_workflow_schema.json +++ b/pkg/parser/schemas/main_workflow_schema.json @@ -12521,10 +12521,10 @@ }, { "type": "array", - "description": "List of failure categories that should trigger issue creation. Categories can be prefixed with '!' to exclude them (e.g., '!inference_access_error'). If only non-prefixed categories are specified, only those categories trigger issues. If only prefixed (excluded) categories are specified, all categories except those trigger issues. If both are specified, categories must match included AND not match excluded. Common categories: agent_failure, timed_out, missing_safe_outputs, report_incomplete, missing_tool, missing_data, inference_access_error, mcp_policy_error, ai_credits_rate_limit_error, max_ai_credits_exceeded, daily_ai_credits_unknown.", + "description": "List of failure categories that should trigger issue creation. Categories can be prefixed with '!' to exclude them (e.g., '!inference_access_error'). If only non-prefixed categories are specified, only those categories trigger issues. If only prefixed (excluded) categories are specified, all categories except those trigger issues. If both are specified, categories must match included AND not match excluded. Common categories: agent_failure, timed_out, transport_wedge, missing_safe_outputs, report_incomplete, missing_tool, missing_data, inference_access_error, mcp_policy_error, ai_credits_rate_limit_error, max_ai_credits_exceeded, daily_ai_credits_unknown.", "items": { "type": "string", - "pattern": "^!?(agent_failure|timed_out|missing_safe_outputs|report_incomplete|engine_driver_failure|safeoutputs_cli_error|invalid_safe_outputs|missing_terminal_safe_output|missing_tool|missing_data|tool_denials_exceeded|cache_miss_misconfiguration|secret_verification_failed|inference_access_error|copilot_org_billing_error|mcp_policy_error|model_not_supported_error|ai_credits_rate_limit_error|unknown_model_ai_credits|max_ai_credits_exceeded|app_token_minting_failed|lockdown_check_failed|stale_lock_file_failed|daily_ai_credits_exceeded|daily_ai_credits_unknown|assignment_errors|assign_copilot_failures|skill_install_failures|create_discussion_errors|code_push_failures|repo_memory_validation_errors|push_repo_memory_failure)$" + "pattern": "^!?(agent_failure|timed_out|transport_wedge|missing_safe_outputs|report_incomplete|engine_driver_failure|safeoutputs_cli_error|invalid_safe_outputs|missing_terminal_safe_output|missing_tool|missing_data|tool_denials_exceeded|cache_miss_misconfiguration|secret_verification_failed|inference_access_error|copilot_org_billing_error|mcp_policy_error|model_not_supported_error|ai_credits_rate_limit_error|unknown_model_ai_credits|max_ai_credits_exceeded|app_token_minting_failed|lockdown_check_failed|stale_lock_file_failed|daily_ai_credits_exceeded|daily_ai_credits_unknown|assignment_errors|assign_copilot_failures|skill_install_failures|create_discussion_errors|code_push_failures|repo_memory_validation_errors|push_repo_memory_failure)$" }, "minItems": 1, "uniqueItems": true