From 19791837197619aec0cba040d6583af549643851 Mon Sep 17 00:00:00 2001 From: yiliang114 Date: Mon, 28 Sep 2026 14:56:58 +0800 Subject: [PATCH 01/27] fix(core): pre-validate bridged tool_call arguments against the target schema The tool_call bridge declaration deliberately types arguments as a bare object so the model-facing schema stays byte-stable across catalog changes, which makes an empty {} envelope-valid even when the deferred target requires fields. resolveDeferredToolCall validated only that envelope and returned the arguments verbatim, so the target's required-field error surfaced post-unwrap in the scheduler as a bare Ajv message (params must have required property 'url') with no target name or remedy; models resubmitted the same empty object until the validation-retry loop guard stopped the turn (LOOP_DETECTED/invalid_tool_params_stagnation). Pre-validate the bridged arguments with the target's own validateToolParams before resolving. A failure now returns an INVALID_TOOL_PARAMS bridge refusal naming the target and the missing field, which the scheduler's existing bridge-refusal accounting already counts toward the retry-loop threshold. The pre-check runs on a structuredClone because SchemaValidator coerces values in place; the scheduler re-validates the returned arguments at build time, so valid calls are unaffected. Fixes #12889 Co-authored-by: Qwen-Coder Patrol-Run: qwen-issue-patrol/jmukuf6832h --- packages/core/src/tools/tool-call.test.ts | 84 +++++++++++++++++++++++ packages/core/src/tools/tool-call.ts | 28 ++++++++ 2 files changed, 112 insertions(+) diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 7d62e73dd53..24d5959cfe9 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -719,4 +719,88 @@ describe('ToolCallTool', () => { ); } }); + + describe('target-schema pre-validation (#12889)', () => { + // Mirrors the issue's web_fetch: both fields required, so `{}` must not + // be accepted just because the bridge envelope types arguments as a + // bare object. + const makeWebFetchLike = () => + new MockTool({ + name: 'web_fetch', + shouldDefer: true, + params: { + type: 'object', + properties: { + url: { type: 'string' }, + prompt: { type: 'string' }, + }, + required: ['url', 'prompt'], + additionalProperties: false, + }, + }); + + it('resolves a call whose arguments satisfy the target schema', async () => { + const target = makeWebFetchLike(); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { + name: 'web_fetch', + arguments: { url: 'https://example.com', prompt: 'summarize' }, + }, + ); + + expect(result).toMatchObject({ + tool: expect.objectContaining({ name: 'web_fetch' }), + arguments: { url: 'https://example.com', prompt: 'summarize' }, + }); + }); + + it('refuses an empty arguments object that misses required target fields', async () => { + // #12889: the bridge validated only its own envelope, so `{}` passed + // and the target's required-field error surfaced post-unwrap as a bare + // Ajv message the model could not act on. The refusal must name the + // target and the missing field. Mutation check: dropping the + // pre-validation in resolveDeferredToolCall turns this red. + const target = makeWebFetchLike(); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: 'web_fetch', arguments: {} }, + ); + + expect(result).toMatchObject({ + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: 'web_fetch', + }); + expect(result).not.toHaveProperty('tool'); + if ('error' in result) { + expect( + result.error.message.startsWith(DEFERRED_TOOL_CALL_REFUSAL_PREFIX), + ).toBe(true); + expect(result.error.message).toContain('"web_fetch"'); + expect(result.error.message).toContain("'url'"); + } + }); + + it('returns the model-sent arguments even when validation coerces a clone', async () => { + // SchemaValidator.validate coerces values in place (numeric strings → + // numbers, etc.). The pre-check must run on a clone: the resolved + // arguments stay exactly what the model sent, and the scheduler + // re-validates them at build time. + const target = new MockTool({ + name: 'counter', + shouldDefer: true, + params: { + type: 'object', + properties: { count: { type: 'integer' } }, + required: ['count'], + }, + }); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: 'counter', arguments: { count: '3' } }, + ); + + expect(result).toMatchObject({ arguments: { count: '3' } }); + }); + }); }); diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 284d631824c..82268205945 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -231,6 +231,34 @@ export async function resolveDeferredToolCall( }; } + // The bridge envelope deliberately types `arguments` as a bare object (the + // declaration must stay byte-stable across catalog changes), so `{}` is + // envelope-valid even when the target requires fields. Pre-validate against + // the target's own schema so the refusal names the target and the missing + // field, instead of surfacing a bare Ajv message after the call has been + // unwrapped (#12889). Validate a clone: SchemaValidator coerces values in + // place, and the scheduler re-validates the returned arguments at build + // time. + let paramsError: string | null = null; + try { + paramsError = target.validateToolParams( + structuredClone(invocation.params.arguments), + ); + } catch { + // A target whose validation throws under this pre-check must not become + // a new bridge failure mode: the scheduler's build() reports the same + // throw as before. + } + if (paramsError) { + return { + error: bridgeRefusal( + `Deferred tool "${target.name}" rejected the arguments: ${paramsError}. Pass arguments matching the schema returned by tool_search for "${target.name}".`, + ), + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: target.name, + }; + } + return { tool: target, arguments: structuredClone(invocation.params.arguments), From 9d6a39dd95eede626e90fe2041d1c2b20a1d9208 Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Mon, 28 Sep 2026 19:01:49 +0800 Subject: [PATCH 02/27] fix(core): exempt media-policy targets from the bridge argument pre-check MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit resolveDeferredToolCall now pre-validates bridged arguments against the target schema (#12889), but omni media-policy targets do not hold their final arguments at that point: both frontends run evaluateMediaPolicyToolCall AFTER bridge resolution (coreToolScheduler before buildInvocation, ACP Session.runTool), and that gate is what resolves resourceId to inputPath and merges defaultArguments/lockedArguments, while BaseMediaPolicyTool's validateToolParams deliberately checks the NATIVE schema rather than the model-visible projection tool_search returned. So the pre-check refused calls the very next stage accepted. A bridged {resourceId, outputDir} call — the shape omni/media-guidance.ts tells the model to send — was refused with "provide exactly one of inputPath ... or resourceId" even though resourceId was provided, and for path-less media the model can never learn the real path, so it cannot repair the call. An operator-pinned modelAccess.lockedArguments.outputDir was unwinnable both ways: projectMediaPolicyToolDeclaration strips the locked key from the model-visible properties AND required, so omitting it failed the pre-check while sending it failed the gate. Each refusal fed recordBatchRetryableToolError and gained RETRY_LOOP_STOP_DIRECTIVE, ending the turn in LOOP_DETECTED / invalid_tool_params_stagnation on a call shape that previously worked. All 14 omniPolicyToolFactories tools inherit the first trigger; the 13 with a native required outputDir also reach the second. Key the exemption off mediaPolicyDescriptor: the code-level fact the gate itself keys off (it passes every non-policy tool through untouched), so the exemption covers exactly the targets whose arguments a downstream stage completes. Nothing fails open — the gate still emits named invalid_params refusals before build(), and build() re-validates the merged arguments against the native schema. Acceptance test added to the existing #12889 block: a target whose model-visible schema omits a key its validateToolParams requires (native required: ['inputPath','outputDir'], projected required: []) resolves with projection-satisfying arguments instead of being refused. Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-closeout/jmul3pj9gc4 --- packages/core/src/tools/tool-call.test.ts | 83 ++++++++++++++++++++++- packages/core/src/tools/tool-call.ts | 36 +++++++--- 2 files changed, 110 insertions(+), 9 deletions(-) diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 24d5959cfe9..f4e5faae95d 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -12,12 +12,13 @@ import { deferredDeclarationFingerprint, type ToolRegistry, } from './tool-registry.js'; -import type { AnyDeclarativeTool } from './tools.js'; +import type { AnyDeclarativeTool, MediaPolicyToolDescriptor } from './tools.js'; import { DEFERRED_TOOL_CALL_REFUSAL_PREFIX, resolveDeferredToolCall, ToolCallTool, } from './tool-call.js'; +import { SchemaValidator } from '../utils/schemaValidator.js'; import { ToolErrorType } from './tool-error.js'; import { ToolNames } from './tool-names.js'; import { DEFAULT_MAX_SUBAGENT_DEPTH } from '../config/config.js'; @@ -802,5 +803,85 @@ describe('ToolCallTool', () => { expect(result).toMatchObject({ arguments: { count: '3' } }); }); + + it('resolves a media-policy target whose arguments the policy gate completes', async () => { + // The projection split a media-policy tool creates: `schema` is the + // model-visible declaration (an operator `modelAccess.lockedArguments` + // key stripped from BOTH properties and required), while + // `validateToolParams` keeps checking the NATIVE schema + // (omni/policy/tools/media-policy-tool.ts). The model is therefore + // correct to omit `outputDir`, and the modelAccess gate — which both + // frontends run AFTER bridge resolution — merges it back in. Running the + // pre-check on these raw arguments refuses a call the next stage accepts, + // and sending the locked key instead makes the gate refuse it: unwinnable + // both ways. Mutation check: dropping the media-policy exemption in + // resolveDeferredToolCall turns this red. + const nativeSchema = { + type: 'object', + properties: { + inputPath: { type: 'string' }, + outputDir: { type: 'string' }, + }, + required: ['inputPath', 'outputDir'], + additionalProperties: false, + }; + const projectedSchema = { + type: 'object', + properties: { inputPath: { type: 'string' } }, + required: [], + additionalProperties: false, + }; + + class MockLockedMediaPolicyTool extends MockTool { + override get mediaPolicyDescriptor(): MediaPolicyToolDescriptor { + return { + kind: 'media_policy', + inputMediaTypes: ['audio'], + outputs: [{ kind: 'media', required: true }], + }; + } + + override get schema() { + return { + name: this.name, + description: this.description, + parametersJsonSchema: projectedSchema, + }; + } + + override validateToolParams(params: { + [key: string]: unknown; + }): string | null { + return SchemaValidator.validate(nativeSchema, params); + } + } + + const target = new MockLockedMediaPolicyTool({ + name: 'omni_transcribe_audio', + shouldDefer: true, + params: nativeSchema, + }); + // The mock really carries the split the defect needs: the model-visible + // schema omits `outputDir`, native validation still requires it. + expect(target.schema.parametersJsonSchema).toEqual(projectedSchema); + expect(target.validateToolParams({ inputPath: '/tmp/in.wav' })).toContain( + "'outputDir'", + ); + + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { + name: 'omni_transcribe_audio', + arguments: { inputPath: '/tmp/in.wav' }, + }, + ); + + expect(result).not.toHaveProperty('error'); + expect(result).not.toHaveProperty('errorType'); + expect(result).toMatchObject({ + tool: expect.objectContaining({ name: 'omni_transcribe_audio' }), + arguments: { inputPath: '/tmp/in.wav' }, + }); + }); }); }); diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 82268205945..b3dcb556a68 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -239,15 +239,35 @@ export async function resolveDeferredToolCall( // unwrapped (#12889). Validate a clone: SchemaValidator coerces values in // place, and the scheduler re-validates the returned arguments at build // time. + // + // Omni media-policy targets are exempt: their arguments are not final here. + // Both frontends run the modelAccess gate AFTER bridge resolution + // (coreToolScheduler's `evaluateMediaPolicyToolCall`, before buildInvocation; + // ACP Session.runTool), and that gate resolves `resourceId` → `inputPath` + // and merges `defaultArguments`/`lockedArguments`. Their `validateToolParams` + // deliberately checks the NATIVE schema rather than the model-visible + // projection `tool_search` returned, so pre-checking the raw bridged + // arguments refuses calls the very next stage accepts — and for an operator + // locked key the refusal is unwinnable both ways (omitting it fails here, + // sending it fails the gate). `mediaPolicyDescriptor` is the code-level fact + // the gate itself keys off (it passes every non-policy tool through + // untouched), so the exemption covers exactly the tools whose arguments a + // downstream stage completes. Nothing fails open: the gate still emits named + // `invalid_params` refusals, and build() re-validates the merged arguments. + const argsCompletedByPolicyGate = + (target as { mediaPolicyDescriptor?: unknown }).mediaPolicyDescriptor !== + undefined; let paramsError: string | null = null; - try { - paramsError = target.validateToolParams( - structuredClone(invocation.params.arguments), - ); - } catch { - // A target whose validation throws under this pre-check must not become - // a new bridge failure mode: the scheduler's build() reports the same - // throw as before. + if (!argsCompletedByPolicyGate) { + try { + paramsError = target.validateToolParams( + structuredClone(invocation.params.arguments), + ); + } catch { + // A target whose validation throws under this pre-check must not become + // a new bridge failure mode: the scheduler's build() reports the same + // throw as before. + } } if (paramsError) { return { From 2c91d42cd73a60fd69eb5d1e0493dcb3eebd1ad9 Mon Sep 17 00:00:00 2001 From: yiliang114 Date: Mon, 28 Sep 2026 23:27:31 +0800 Subject: [PATCH 03/27] fix(core): order the bridge argument pre-check behind truncation and policy gates MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round-2 review follow-up on the #12889 pre-validation: - Yield the pre-check when the request is stamped wasOutputTruncated, so a max_tokens-cut bridged envelope surfaces as truncation (the scheduler's existing truncation guards own the message) instead of a schema mismatch. - Consult the caller's per-target execution policy (scheduler execution allowlist / ACP permission-manager enablement) before validating arguments, so a denied target keeps its specific EXECUTION_DENIED. - Media-policy targets now pre-check against the model-visible projection (target.schema) instead of skipping validation: a missing model-visible required field is refused naming the target, while gate-completed arguments (resourceId handles, operator-locked keys) stay unrefusable at the bridge. - Plumb the validated targetName through bridgeResolutionError and key per-tool retry accounting on it (widening the batch-start prune to match), so alternating failures against distinct targets accrue per-target and reach the retry-loop threshold. - Coalesce AgentTool.refreshSubagents kicks onto the in-flight refresh — validation now fires twice per bridged Agent call — while the change listener re-arms one follow-up so a mid-refresh change is not dropped. - Read mediaPolicyDescriptor typed instead of through a structural cast. - Tests pin: the resourceId trigger shape, a projector-derived projection fixture, the throwing-validator fall-through, and scheduler coverage for the truncation / policy-denial / per-target-accounting orderings. Co-authored-by: Qwen-Coder Patrol-Run: qwen-issue-patrol/jmulbkgfd2o --- .../src/acp-integration/session/Session.ts | 9 + .../core/src/core/coreToolScheduler.test.ts | 242 +++++++++++++++++- packages/core/src/core/coreToolScheduler.ts | 30 ++- packages/core/src/tools/agent/agent.test.ts | 29 +++ packages/core/src/tools/agent/agent.ts | 38 ++- packages/core/src/tools/tool-call.test.ts | 217 ++++++++++++---- packages/core/src/tools/tool-call.ts | 75 ++++-- 7 files changed, 561 insertions(+), 79 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.ts b/packages/cli/src/acp-integration/session/Session.ts index bd01ae16b50..8dd4505e8dc 100644 --- a/packages/cli/src/acp-integration/session/Session.ts +++ b/packages/cli/src/acp-integration/session/Session.ts @@ -13408,6 +13408,15 @@ export class Session implements SessionContext { // all three frontends (wenshao triage follow-up). Omitting it would // fail closed, not open, but the corner case should agree everywhere. maxSubagentDepth: this.config.getMaxSubagentDepth(), + // The L1 enablement gate below runs after resolution; consult the + // same policy inside resolution so a denied target keeps its + // EXECUTION_DENIED instead of a parameter pre-check refusal. + ...(pm + ? { + isTargetExecutionAllowed: (targetName: string) => + pm.isToolEnabled(targetName), + } + : {}), }); const bridgeCancellation = cancelBeforeExecutionIfAborted(toolName); if (bridgeCancellation) return bridgeCancellation; diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index a5284cac66f..31581fad37b 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -1584,6 +1584,158 @@ describe('CoreToolScheduler', () => { } }); + it('denies a policy-blocked bridged target before validating its arguments', async () => { + // The owner execution allowlist must win over the bridge argument + // pre-check: a denied target gets the specific EXECUTION_DENIED naming + // the policy, not an INVALID_TOOL_PARAMS parameter error for a call that + // could never run (and the denied tool's validator never executes). The + // allowlist test above does not discriminate — its MockTool carries no + // required schema, so the pre-check passes it. Mutation check: running + // the pre-check ahead of the policy gate turns this red with "params + // must have required property 'url'". + const execute = vi.fn(); + const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); + const deferred = new MockTool({ + name: 'web_fetch', + shouldDefer: true, + params: { + type: 'object', + properties: { + url: { type: 'string' }, + prompt: { type: 'string' }, + }, + required: ['url', 'prompt'], + additionalProperties: false, + }, + execute, + }); + const { scheduler, onAllToolCallsComplete } = + createSchedulerForLegacyToolTests({ + toolsByName: new Map([ + [bridge.name, bridge], + [deferred.name, deferred], + ]), + deferredHiddenNames: new Set([deferred.name]), + isToolExecutionAllowed: (name: string) => name !== 'web_fetch', + }); + + await scheduler.schedule( + { + callId: 'bridge-deny-before-precheck', + name: ToolNames.TOOL_CALL, + args: { name: deferred.name, arguments: {} }, + isClientInitiated: false, + prompt_id: 'prompt-bridge-deny-before-precheck', + }, + new AbortController().signal, + ); + + expect(execute).not.toHaveBeenCalled(); + const completed = onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.errorType).toBe(ToolErrorType.EXECUTION_DENIED); + expect(completed.response.error?.message).toContain( + "is not permitted by this agent's tool policy", + ); + expect(completed.response.error?.message).not.toContain( + "required property 'url'", + ); + } + }); + + it('accrues bridge argument refusals per target for retry-loop detection', async () => { + // The refusal carries the validated targetName for exactly this + // accounting: alternating broken bridged calls against two distinct + // targets must strike per-target counters. Keyed on the wrapper + // (`tool_call`) instead, each strike prunes the other target's counter + // and neither ever reaches VALIDATION_RETRY_LOOP_THRESHOLD — the loop + // this PR exists to stop escapes the early stop directive. Mutation + // check: keying the refusal branch on reqInfo.name turns this red. + const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); + const makeDeferred = (name: string, required: string[]) => + new MockTool({ + name, + shouldDefer: true, + params: { + type: 'object', + properties: Object.fromEntries( + required.map((key) => [key, { type: 'string' }]), + ), + required, + additionalProperties: false, + }, + }); + const writeFile = makeDeferred('write_file', ['file_path', 'content']); + const webFetch = makeDeferred('web_fetch', ['url', 'prompt']); + const { scheduler, onAllToolCallsComplete } = + createSchedulerForLegacyToolTests({ + toolsByName: new Map([ + [bridge.name, bridge], + [writeFile.name, writeFile], + [webFetch.name, webFetch], + ]), + deferredHiddenNames: new Set([writeFile.name, webFetch.name]), + }); + + const scheduleAlternatingBatch = async (batchId: number) => { + onAllToolCallsComplete.mockClear(); + await scheduler.schedule( + [ + { + callId: `bridge-loop-${batchId}-write`, + name: ToolNames.TOOL_CALL, + args: { name: 'write_file', arguments: {} }, + isClientInitiated: false, + prompt_id: 'prompt-bridge-loop', + }, + { + callId: `bridge-loop-${batchId}-fetch`, + name: ToolNames.TOOL_CALL, + args: { name: 'web_fetch', arguments: {} }, + isClientInitiated: false, + prompt_id: 'prompt-bridge-loop', + }, + ], + new AbortController().signal, + ); + await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); + return onAllToolCallsComplete.mock.calls[0][0] as ToolCall[]; + }; + + for (const batch of [ + await scheduleAlternatingBatch(1), + await scheduleAlternatingBatch(2), + ]) { + expect(batch).toHaveLength(2); + for (const completed of batch) { + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.errorType).toBe( + ToolErrorType.INVALID_TOOL_PARAMS, + ); + expect(completed.response.error?.message).not.toContain( + 'RETRY LOOP DETECTED', + ); + } + } + } + + const third = await scheduleAlternatingBatch(3); + expect(third).toHaveLength(2); + for (const completed of third) { + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.errorType).toBe( + ToolErrorType.INVALID_TOOL_PARAMS, + ); + expect(completed.response.error?.message).toContain( + 'RETRY LOOP DETECTED', + ); + } + } + }); + it('applies the retry-loop directive to repeated invalid tool_call envelopes', async () => { const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); const { scheduler, onAllToolCallsComplete } = @@ -11182,16 +11334,29 @@ describe('CoreToolScheduler truncated output protection', () => { function createTruncationTestScheduler( tool: AnyDeclarativeTool, toolNames: string[], + options: { + extraTools?: AnyDeclarativeTool[]; + deferredHiddenNames?: ReadonlySet; + } = {}, ) { const onAllToolCallsComplete = vi.fn(); const onToolCallsUpdate = vi.fn(); + const toolsByName = new Map([ + [tool.name, tool], + ...(options.extraTools ?? []).map((extra) => [extra.name, extra] as const), + ]); const mockToolRegistry = { - getTool: () => tool, - ensureTool: async () => tool, - getAllToolNames: () => toolNames, + getTool: (name: string) => toolsByName.get(name) ?? tool, + ensureTool: async (name: string) => toolsByName.get(name) ?? tool, + getAllToolNames: () => [ + ...toolNames, + ...(options.extraTools ?? []).map((extra) => extra.name), + ], getFunctionDeclarations: () => [], tools: new Map(), + isDeferredAndHidden: (name: string) => + options.deferredHiddenNames?.has(name) ?? false, } as unknown as ToolRegistry; const mockConfig = { @@ -11222,6 +11387,7 @@ describe('CoreToolScheduler truncated output protection', () => { isInteractive: () => true, getMessageBus: vi.fn().mockReturnValue(undefined), getDisableAllHooks: vi.fn().mockReturnValue(true), + getMaxSubagentDepth: () => DEFAULT_MAX_SUBAGENT_DEPTH, } as unknown as Config; const scheduler = new CoreToolScheduler({ @@ -11402,6 +11568,76 @@ describe('CoreToolScheduler truncated output protection', () => { } }); + it('should prefer truncation handling over the bridge argument pre-check for a bridged write_file call', async () => { + // Same as the direct-call case above, but reached through the tool_call + // bridge (#12889): the bridge pre-check runs ahead of the truncation + // guards below, so it must yield when the request is stamped + // wasOutputTruncated — otherwise a max_tokens-cut envelope surfaces as a + // schema mismatch and the model re-sends the same oversized write. + // Mutation check: dropping the wasOutputTruncated condition from the + // pre-check turns this red with "params must have required property + // 'content'". + const writeFileConfig = { + getProjectRoot: () => '/tmp', + getTargetDir: () => '/tmp', + getFileSystemService: () => ({ + readTextFile: vi.fn(), + writeTextFile: vi.fn(), + }), + getDefaultFileEncoding: () => undefined, + setApprovalMode: vi.fn(), + } as unknown as Config; + const writeFileTool = new WriteFileTool(writeFileConfig); + const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); + const toolSearch = new MockTool({ name: ToolNames.TOOL_SEARCH }); + const { scheduler, onAllToolCallsComplete } = createTruncationTestScheduler( + writeFileTool, + [WriteFileTool.Name], + { + extraTools: [bridge, toolSearch], + deferredHiddenNames: new Set([WriteFileTool.Name]), + }, + ); + + await scheduler.schedule( + [ + { + callId: '1', + name: ToolNames.TOOL_CALL, + args: { + name: WriteFileTool.Name, + arguments: { file_path: '/tmp/test.txt' }, + }, + isClientInitiated: false, + prompt_id: 'prompt-id-bridge-write-file-truncated', + wasOutputTruncated: true, + }, + ], + new AbortController().signal, + ); + + await vi.waitFor(() => { + expect(onAllToolCallsComplete).toHaveBeenCalled(); + }); + + const completedCalls = onAllToolCallsComplete.mock + .calls[0][0] as ToolCall[]; + expect(completedCalls).toHaveLength(1); + const completedCall = completedCalls[0]; + expect(completedCall.status).toBe('error'); + + if (completedCall.status === 'error') { + const errorMessage = completedCall.response.error?.message; + expect(errorMessage).toContain('truncated due to max_tokens limit'); + expect(errorMessage).toContain( + 'rejected to prevent writing truncated content', + ); + expect(errorMessage).not.toContain( + "params must have required property 'content'", + ); + } + }); + it('should inject retry loop directive after repeated truncated write_file rejections', async () => { const writeFileConfig = { getProjectRoot: () => '/tmp', diff --git a/packages/core/src/core/coreToolScheduler.ts b/packages/core/src/core/coreToolScheduler.ts index c5ed0c542a7..fdb7b993183 100644 --- a/packages/core/src/core/coreToolScheduler.ts +++ b/packages/core/src/core/coreToolScheduler.ts @@ -1073,6 +1073,12 @@ type SchedulerToolCallRequestInfo = ToolCallRequestInfo & { bridgeResolutionError?: { error: Error; type: ToolErrorType; + /** + * Validated bridge target for per-tool parameter-error accounting (the + * wrapper name `tool_call` cannot distinguish targets). Absent when the + * refusal never reached a validated target (e.g. unknown tool). + */ + targetName?: string; }; }; @@ -2710,6 +2716,12 @@ export class CoreToolScheduler { resolveDeferredToolCall(this.toolRegistry, request.args, { // Match prepareTools's depth-gated AgentTool policy. maxSubagentDepth: this.config.getMaxSubagentDepth(), + // A truncated response's arguments are incomplete for transport + // reasons: the pre-check must yield to the truncation guards below. + wasOutputTruncated: request.wasOutputTruncated, + // The owner policy must win over the argument pre-check so a denied + // target keeps its specific EXECUTION_DENIED. + isTargetExecutionAllowed: this.isToolExecutionAllowed, }), ); if ('error' in resolution) { @@ -2718,6 +2730,7 @@ export class CoreToolScheduler { bridgeResolutionError: { error: resolution.error, type: resolution.errorType, + targetName: resolution.targetName, }, }; } @@ -2896,9 +2909,17 @@ export class CoreToolScheduler { // present in the current batch. Keeping every tracked tool's counters // whenever any current request matched caused stale counts for // unrelated tools to survive and fire RETRY LOOP DETECTED prematurely - // the next time those tools were used. + // the next time those tools were used. A refused bridge request keeps + // the wrapper name (`tool_call`), so its validated target name must + // join the presence set alongside it. if (this.validationRetryCounts.size > 0) { - const currentToolNames = new Set(requestsToProcess.map((r) => r.name)); + const currentToolNames = new Set( + requestsToProcess.flatMap((r) => + r.bridgeResolutionError?.targetName !== undefined + ? [r.name, r.bridgeResolutionError.targetName] + : [r.name], + ), + ); for (const key of [...this.validationRetryCounts.keys()]) { const sep = key.indexOf(':'); const toolName = sep === -1 ? key : key.slice(0, sep); @@ -2978,8 +2999,11 @@ export class CoreToolScheduler { reqInfo.bridgeResolutionError.type === ToolErrorType.INVALID_TOOL_PARAMS ) { + // Key on the validated target, not the `tool_call` wrapper: + // alternating failures against distinct targets must accrue + // per-target instead of pruning each other's counter. const count = recordBatchRetryableToolError( - reqInfo.name, + reqInfo.bridgeResolutionError.targetName ?? reqInfo.name, bridgeError.message, ); if (count >= VALIDATION_RETRY_LOOP_THRESHOLD) { diff --git a/packages/core/src/tools/agent/agent.test.ts b/packages/core/src/tools/agent/agent.test.ts index 803bd8a748a..5c450ca819e 100644 --- a/packages/core/src/tools/agent/agent.test.ts +++ b/packages/core/src/tools/agent/agent.test.ts @@ -1456,6 +1456,35 @@ describe('AgentTool', () => { expect(result).toBeNull(); }); + it('coalesces a second synchronous refresh kick onto the in-flight refresh', async () => { + // An unknown subagent_type kicks refreshSubagents so the cache catches + // up — but validation can fire twice for one call (the tool_call + // bridge pre-check validates ahead of build(), #12889), and each kick + // used to start its own full subagent rescan + llmClient.setTools. A + // kick arriving while a refresh is in flight must coalesce onto it. + await agentTool.refreshSubagents(); + const listSpy = vi.mocked(mockSubagentManager.listSubagents); + listSpy.mockClear(); + + agentTool.validateToolParams({ + ...validParams, + subagent_type: 'missing', + }); + agentTool.validateToolParams({ + ...validParams, + subagent_type: 'missing', + }); + expect(listSpy).toHaveBeenCalledTimes(1); + + // After the in-flight refresh settles, a later kick re-scans. + await vi.runAllTimersAsync(); + agentTool.validateToolParams({ + ...validParams, + subagent_type: 'missing', + }); + expect(listSpy).toHaveBeenCalledTimes(2); + }); + it('should reject empty description', async () => { const result = agentTool.validateToolParams({ ...validParams, diff --git a/packages/core/src/tools/agent/agent.ts b/packages/core/src/tools/agent/agent.ts index a2f659db3f7..e63729ae229 100644 --- a/packages/core/src/tools/agent/agent.ts +++ b/packages/core/src/tools/agent/agent.ts @@ -886,7 +886,7 @@ export class AgentTool extends BaseDeclarativeTool { this.delegationSurface = resolveAgentDelegationSurface(config); this.subagentManager = config.getSubagentManager(); this.removeChangeListener = this.subagentManager.addChangeListener(() => { - void this.refreshSubagents(); + this.refreshSubagentsFromListener(); }); // Initialize the tool asynchronously @@ -897,11 +897,45 @@ export class AgentTool extends BaseDeclarativeTool { this.removeChangeListener(); } + private refreshInFlight: Promise | undefined; + private listenerRefreshArmed = false; + /** * Asynchronously initializes the tool by loading available subagents * and updating the description and schema. + * + * Concurrent kicks coalesce onto the in-flight refresh: validateToolParams + * fires one per validated call (and the bridge argument pre-check doubles + * that for a bridged Agent call), and a second scan + setTools mid-turn + * buys nothing — the in-flight scan already reads the current state. + */ + refreshSubagents(): Promise { + this.refreshInFlight ??= this.runRefreshSubagents().finally(() => { + this.refreshInFlight = undefined; + }); + return this.refreshInFlight; + } + + /** + * The change listener must not be dropped like a validation kick: a change + * landing mid-refresh is not reflected in the in-flight scan, so arm one + * follow-up refresh instead of coalescing into the stale read. */ - async refreshSubagents(): Promise { + private refreshSubagentsFromListener(): void { + if (this.refreshInFlight) { + if (!this.listenerRefreshArmed) { + this.listenerRefreshArmed = true; + void this.refreshInFlight.finally(() => { + this.listenerRefreshArmed = false; + void this.refreshSubagents(); + }); + } + return; + } + void this.refreshSubagents(); + } + + private async runRefreshSubagents(): Promise { try { this.availableSubagents = await this.subagentManager.listSubagents(); this.updateDescriptionAndSchema(); diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index f4e5faae95d..96a0e3a3d9c 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -19,6 +19,11 @@ import { ToolCallTool, } from './tool-call.js'; import { SchemaValidator } from '../utils/schemaValidator.js'; +import { projectMediaPolicyToolDeclaration } from '../omni/policy/model-access.js'; +import { + validateMediaPolicyIoParams, + type MediaPolicyIoParams, +} from '../omni/policy/tools/media-policy-tool.js'; import { ToolErrorType } from './tool-error.js'; import { ToolNames } from './tool-names.js'; import { DEFAULT_MAX_SUBAGENT_DEPTH } from '../config/config.js'; @@ -804,66 +809,98 @@ describe('ToolCallTool', () => { expect(result).toMatchObject({ arguments: { count: '3' } }); }); - it('resolves a media-policy target whose arguments the policy gate completes', async () => { - // The projection split a media-policy tool creates: `schema` is the - // model-visible declaration (an operator `modelAccess.lockedArguments` - // key stripped from BOTH properties and required), while - // `validateToolParams` keeps checking the NATIVE schema - // (omni/policy/tools/media-policy-tool.ts). The model is therefore - // correct to omit `outputDir`, and the modelAccess gate — which both - // frontends run AFTER bridge resolution — merges it back in. Running the - // pre-check on these raw arguments refuses a call the next stage accepts, - // and sending the locked key instead makes the gate refuse it: unwinnable - // both ways. Mutation check: dropping the media-policy exemption in - // resolveDeferredToolCall turns this red. - const nativeSchema = { - type: 'object', - properties: { - inputPath: { type: 'string' }, - outputDir: { type: 'string' }, - }, - required: ['inputPath', 'outputDir'], - additionalProperties: false, - }; - const projectedSchema = { - type: 'object', - properties: { inputPath: { type: 'string' } }, - required: [], - additionalProperties: false, - }; + // The shared native shape of the omni media-policy family: io params + // with `resourceId` as the model-facing `inputPath` alternative, and + // only `outputDir` required natively (0 of the 14 shipped tools require + // inputPath — the call gate resolves resourceId → inputPath before + // build() validates). + const mediaPolicyNativeSchema = { + type: 'object', + properties: { + inputPath: { type: 'string' }, + resourceId: { type: 'string' }, + outputDir: { type: 'string' }, + }, + required: ['outputDir'], + additionalProperties: false, + }; - class MockLockedMediaPolicyTool extends MockTool { - override get mediaPolicyDescriptor(): MediaPolicyToolDescriptor { - return { - kind: 'media_policy', - inputMediaTypes: ['audio'], - outputs: [{ kind: 'media', required: true }], - }; - } + // A MockTool carrying the real media-policy split: `schema` is derived + // through the production projector (locked keys stripped from properties + // AND required) instead of a hand-written literal, while + // `validateToolParams` keeps checking the NATIVE schema plus the io + // value rule, exactly like BaseMediaPolicyTool + // (omni/policy/tools/media-policy-tool.ts). + class MockMediaPolicyTool extends MockTool { + constructor(private readonly lockedArguments: Record) { + super({ + name: 'omni_transcribe_audio', + shouldDefer: true, + params: mediaPolicyNativeSchema, + }); + } + + override get mediaPolicyDescriptor(): MediaPolicyToolDescriptor { + return { + kind: 'media_policy', + inputMediaTypes: ['audio'], + outputs: [{ kind: 'media', required: true }], + }; + } - override get schema() { - return { + override get schema() { + return projectMediaPolicyToolDeclaration( + { + getOmniPolicyToolsSettings: () => ({ + [this.name]: { + modelAccess: { + enabled: true, + lockedArguments: this.lockedArguments, + }, + }, + }), + }, + { name: this.name, description: this.description, - parametersJsonSchema: projectedSchema, - }; - } + parametersJsonSchema: mediaPolicyNativeSchema, + }, + ); + } - override validateToolParams(params: { - [key: string]: unknown; - }): string | null { - return SchemaValidator.validate(nativeSchema, params); - } + override validateToolParams(params: { + [key: string]: unknown; + }): string | null { + return ( + SchemaValidator.validate(mediaPolicyNativeSchema, params) ?? + validateMediaPolicyIoParams(params as unknown as MediaPolicyIoParams) + ); } + } - const target = new MockLockedMediaPolicyTool({ - name: 'omni_transcribe_audio', - shouldDefer: true, - params: nativeSchema, - }); - // The mock really carries the split the defect needs: the model-visible - // schema omits `outputDir`, native validation still requires it. - expect(target.schema.parametersJsonSchema).toEqual(projectedSchema); + it('resolves a media-policy target whose arguments the policy gate completes', async () => { + // The projection split a media-policy tool creates: `schema` is the + // model-visible declaration (an operator `modelAccess.lockedArguments` + // key stripped from BOTH properties and required), while + // `validateToolParams` keeps checking the NATIVE schema. The model is + // therefore correct to omit `outputDir`, and the modelAccess gate — + // which both frontends run AFTER bridge resolution — merges it back + // in. Pre-checking the raw arguments against the native schema refuses + // a call the next stage accepts, and sending the locked key instead + // makes the gate refuse it: unwinnable both ways. Mutation check: + // dropping the media-policy branch in resolveDeferredToolCall turns + // this red. + const target = new MockMediaPolicyTool({ outputDir: '/locked/out' }); + // The mock really carries the split the defect needs, derived through + // the real projector rather than pinned as a literal: the locked key + // leaves properties and required, the rest of the surface stays. + const projection = target.schema.parametersJsonSchema as { + properties: Record; + required?: string[]; + }; + expect(projection.properties).toHaveProperty('inputPath'); + expect(projection.properties).not.toHaveProperty('outputDir'); + expect(projection.required ?? []).not.toContain('outputDir'); expect(target.validateToolParams({ inputPath: '/tmp/in.wav' })).toContain( "'outputDir'", ); @@ -883,5 +920,79 @@ describe('ToolCallTool', () => { arguments: { inputPath: '/tmp/in.wav' }, }); }); + + it('resolves a media-policy target called with a resourceId handle and no inputPath', async () => { + // The call shape omni/media-guidance.ts instructs the model to send: + // an opaque session media handle instead of inputPath (the gate + // resolves it to inputPath AFTER bridge resolution), with the locked + // outputDir omitted. The bridge must not apply the native schema or + // the io value rule here — both demand fields only the gate supplies. + // Mutation check: removing the media-policy branch, or pre-checking + // the native schema, refuses this on the locked outputDir. + const target = new MockMediaPolicyTool({ outputDir: '/locked/out' }); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { + name: 'omni_transcribe_audio', + arguments: { resourceId: 'media-1-abcd' }, + }, + ); + + expect(result).not.toHaveProperty('error'); + expect(result).not.toHaveProperty('errorType'); + expect(result).toMatchObject({ + tool: expect.objectContaining({ name: 'omni_transcribe_audio' }), + arguments: { resourceId: 'media-1-abcd' }, + }); + }); + + it('refuses a media-policy target whose arguments miss a model-visible required field', async () => { + // With no lockedArguments the projection is the native schema, so a + // bridged `{}` must still be refused here — naming the target and the + // missing field — instead of surfacing a bare Ajv message from build() + // under the wrapper name. Mutation check: skipping validation for + // media-policy targets turns this red. + const target = new MockMediaPolicyTool({}); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: 'omni_transcribe_audio', arguments: {} }, + ); + + expect(result).toMatchObject({ + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: 'omni_transcribe_audio', + }); + if ('error' in result) { + expect(result.error.message).toContain('"omni_transcribe_audio"'); + expect(result.error.message).toContain("'outputDir'"); + } + }); + + it('resolves a target whose validateToolParams throws, leaving the throw to build()', async () => { + // A throwing validator must not become a new bridge failure mode: the + // scheduler's build() reports the same throw as before. Mutation + // check: dropping the try/catch around the pre-check turns this red. + class ThrowingValidatorTool extends MockTool { + override validateToolParams(): string | null { + throw new Error('boom from validator'); + } + } + const target = new ThrowingValidatorTool({ + name: 'throwing_tool', + shouldDefer: true, + params: { type: 'object', properties: {} }, + }); + + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: 'throwing_tool', arguments: {} }, + ); + + expect(result).not.toHaveProperty('error'); + expect(result).toMatchObject({ + tool: expect.objectContaining({ name: 'throwing_tool' }), + arguments: {}, + }); + }); }); }); diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index b3dcb556a68..23ab66a5be5 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -21,6 +21,7 @@ import { deferredDeclarationFingerprint, type ToolRegistry, } from './tool-registry.js'; +import { SchemaValidator } from '../utils/schemaValidator.js'; import { getExcludedToolUnavailableMessage, getLeaderOnlyToolUnavailableMessage, @@ -51,6 +52,24 @@ export type DeferredToolCallResolution = export interface DeferredToolCallOptions { /** Omission keeps AgentTool excluded in subagent contexts. */ maxSubagentDepth?: number; + /** + * Set when the model's response was cut by max_tokens. The bridged + * arguments may be incomplete for transport reasons rather than a schema + * misreading, so the argument pre-check must yield to the caller's + * truncation-aware handling (the scheduler rejects truncated Edit-kind + * calls outright and appends truncation guidance to build-time validation + * failures) instead of reporting a schema mismatch. + */ + wasOutputTruncated?: boolean; + /** + * The caller's per-target execution policy (scheduler execution allowlist + * / permission-manager enablement). Consulted before the argument + * pre-check so a denied target keeps its specific EXECUTION_DENIED refusal + * instead of surfacing a parameter error for a call that could never run. + */ + isTargetExecutionAllowed?: ( + targetName: string, + ) => boolean | Promise; } export const DEFERRED_TOOL_CALL_REFUSAL_PREFIX = '[tool_call bridge refused] '; @@ -180,6 +199,21 @@ export async function resolveDeferredToolCall( errorType: ToolErrorType.EXECUTION_DENIED, }; } + // The caller's execution policy precedes the hidden-tool gate and the + // argument pre-check: a denied target keeps its specific denial rather + // than a parameter error for a call that could never run. + if ( + options?.isTargetExecutionAllowed !== undefined && + !(await options.isTargetExecutionAllowed(target.name)) + ) { + return { + error: bridgeRefusal( + `Tool "${target.name}" is not permitted by this agent's tool policy (execution allowlist or disallowedTools blocklist).`, + ), + errorType: ToolErrorType.EXECUTION_DENIED, + targetName: target.name, + }; + } if (!registry.isDeferredAndHidden(target.name)) { return { @@ -240,29 +274,34 @@ export async function resolveDeferredToolCall( // place, and the scheduler re-validates the returned arguments at build // time. // - // Omni media-policy targets are exempt: their arguments are not final here. - // Both frontends run the modelAccess gate AFTER bridge resolution + // Omni media-policy targets are pre-checked against the model-visible + // projection (their `schema` getter) instead of the native schema: both + // frontends run the modelAccess gate AFTER bridge resolution // (coreToolScheduler's `evaluateMediaPolicyToolCall`, before buildInvocation; // ACP Session.runTool), and that gate resolves `resourceId` → `inputPath` // and merges `defaultArguments`/`lockedArguments`. Their `validateToolParams` - // deliberately checks the NATIVE schema rather than the model-visible - // projection `tool_search` returned, so pre-checking the raw bridged - // arguments refuses calls the very next stage accepts — and for an operator - // locked key the refusal is unwinnable both ways (omitting it fails here, - // sending it fails the gate). `mediaPolicyDescriptor` is the code-level fact - // the gate itself keys off (it passes every non-policy tool through - // untouched), so the exemption covers exactly the tools whose arguments a - // downstream stage completes. Nothing fails open: the gate still emits named - // `invalid_params` refusals, and build() re-validates the merged arguments. - const argsCompletedByPolicyGate = - (target as { mediaPolicyDescriptor?: unknown }).mediaPolicyDescriptor !== - undefined; + // deliberately checks the NATIVE schema plus io value rules that assume + // that completion, so running it here refuses calls the very next stage + // accepts — and for an operator locked key the refusal is unwinnable both + // ways (omitting it fails natively, sending it fails the gate). The + // projection has locked keys stripped from `required`, so validating it + // still refuses a missing model-visible required field while never + // demanding a key the model is forbidden to send. `mediaPolicyDescriptor` + // is the code-level fact the gate itself keys off (it passes every + // non-policy tool through untouched), so the narrowed check covers exactly + // the tools whose arguments a downstream stage completes. Nothing fails + // open: the gate still emits named `invalid_params` refusals, and build() + // re-validates the merged arguments against the native schema. + const isMediaPolicyTarget = target.mediaPolicyDescriptor !== undefined; let paramsError: string | null = null; - if (!argsCompletedByPolicyGate) { + // A truncated response yields to the caller's truncation handling: the + // arguments are incomplete for transport reasons, not a schema misreading. + if (!options?.wasOutputTruncated) { try { - paramsError = target.validateToolParams( - structuredClone(invocation.params.arguments), - ); + const argsClone = structuredClone(invocation.params.arguments); + paramsError = isMediaPolicyTarget + ? SchemaValidator.validate(target.schema.parametersJsonSchema, argsClone) + : target.validateToolParams(argsClone); } catch { // A target whose validation throws under this pre-check must not become // a new bridge failure mode: the scheduler's build() reports the same From 191879b7a85c7f57585d882c24abe62e776f3e03 Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Mon, 28 Sep 2026 23:57:34 +0800 Subject: [PATCH 04/27] style(core): apply prettier formatting flagged by the lint gate The Lint & Static job failed at the "Run Prettier" step on packages/core/src/core/coreToolScheduler.test.ts and packages/core/src/tools/tool-call.ts. Reformat both with the repository's pinned prettier (3.6.1); no semantic change. Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-conflict/jmulf520ncq --- packages/core/src/core/coreToolScheduler.test.ts | 4 +++- packages/core/src/tools/tool-call.ts | 9 +++++---- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index 31581fad37b..7b0affe8692 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -11344,7 +11344,9 @@ describe('CoreToolScheduler truncated output protection', () => { const toolsByName = new Map([ [tool.name, tool], - ...(options.extraTools ?? []).map((extra) => [extra.name, extra] as const), + ...(options.extraTools ?? []).map( + (extra) => [extra.name, extra] as const, + ), ]); const mockToolRegistry = { getTool: (name: string) => toolsByName.get(name) ?? tool, diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 23ab66a5be5..641368e6300 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -67,9 +67,7 @@ export interface DeferredToolCallOptions { * pre-check so a denied target keeps its specific EXECUTION_DENIED refusal * instead of surfacing a parameter error for a call that could never run. */ - isTargetExecutionAllowed?: ( - targetName: string, - ) => boolean | Promise; + isTargetExecutionAllowed?: (targetName: string) => boolean | Promise; } export const DEFERRED_TOOL_CALL_REFUSAL_PREFIX = '[tool_call bridge refused] '; @@ -300,7 +298,10 @@ export async function resolveDeferredToolCall( try { const argsClone = structuredClone(invocation.params.arguments); paramsError = isMediaPolicyTarget - ? SchemaValidator.validate(target.schema.parametersJsonSchema, argsClone) + ? SchemaValidator.validate( + target.schema.parametersJsonSchema, + argsClone, + ) : target.validateToolParams(argsClone); } catch { // A target whose validation throws under this pre-check must not become From 4c7ebf961cc6126e0e26c8139a365b238c645753 Mon Sep 17 00:00:00 2001 From: yiliang114 Date: Tue, 29 Sep 2026 08:45:58 +0800 Subject: [PATCH 05/27] fix(core): close R3 review gaps in bridge pre-check policy, retry accounting, and agent refresh MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round-3 review follow-ups for the tool_call bridge argument pre-check: - Scheduler: forward the permission-manager policy as a pre-check suppression so a pm-denied bridged target surfaces the loop's richer EXECUTION_DENIED (deny-rule attribution) instead of an INVALID_TOOL_PARAMS refusal for a call that could never run (R3-3). - Scheduler: drop the post-resolution isToolExecutionAllowed gate — resolution already consults the same constructor-fixed predicate on the same target name, so the copy was unreachable and free to drift (R3-2). - Scheduler: widen the retry-counter presence set only for INVALID_TOOL_PARAMS bridge refusals (the one error type that accrues), so an EXECUTION_DENIED batch no longer preserves stale counters (R3-4), and record bridged pre-check refusals in a channel-marked namespace so they no longer prefix-prune (or get pruned by) the same target's direct validation failures every mixed batch (R3-5). - tool-call: narrow the pre-check to the target's model-visible schema (SchemaValidator) for all targets, so value-level rules (fs stats, content scans, the AgentTool refresh kick) run exactly once at build() time instead of twice per bridged call (R3-12). - AgentTool: route the unknown-subagent_type validation kick through the listener's arm-once path so a kick landing mid-scan queues one follow-up instead of being dropped (R3-6), guard the armed follow-up against post-dispose execution (R3-8), and catch the armed chain per this file's void-boundary contract (R3-9). Tests: pin the ACP policy-denial path (R3-1), the listener mid-refresh arm (R3-7), the policy-gate/hidden-gate ordering (R3-10), and each behavior fix above with its own mutation-checked case. Co-authored-by: Qwen-Coder Patrol-Run: qwen-issue-patrol/jmuluuwer2v --- .../acp-integration/session/Session.test.ts | 108 +++++++ .../core/src/core/coreToolScheduler.test.ts | 264 +++++++++++++++++- packages/core/src/core/coreToolScheduler.ts | 76 +++-- packages/core/src/tools/agent/agent.test.ts | 128 ++++++++- packages/core/src/tools/agent/agent.ts | 51 ++-- packages/core/src/tools/tool-call.test.ts | 85 +++++- packages/core/src/tools/tool-call.ts | 66 +++-- 7 files changed, 687 insertions(+), 91 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.test.ts b/packages/cli/src/acp-integration/session/Session.test.ts index b2551612035..218f8273ba4 100644 --- a/packages/cli/src/acp-integration/session/Session.test.ts +++ b/packages/cli/src/acp-integration/session/Session.test.ts @@ -18031,6 +18031,114 @@ describe('Session', () => { ); }); + it('keeps EXECUTION_DENIED for a policy-denied bridge target instead of a parameter pre-check refusal', async () => { + // The ACP half of the resolution-time policy wiring: with the + // wrapper allowed but the target denied, resolution must refuse with + // the policy denial BEFORE the argument pre-check — otherwise a + // denied target with malformed arguments gets INVALID_TOOL_PARAMS + // plus an invalid-parameter strike toward the loop stop for a call + // that could never run. Mutation check: removing the + // `...(pm ? { isTargetExecutionAllowed } : {})` spread in + // Session.runTool turns this red (errorType flips to + // INVALID_TOOL_PARAMS and invalidToolParamErrors gains an entry). + mockConfig.getApprovalMode = vi.fn().mockReturnValue(ApprovalMode.YOLO); + mockConfig.getPermissionManager = vi.fn().mockReturnValue({ + isToolEnabled: vi.fn(async (name: string) => name !== 'web_fetch'), + }); + const bridge = { + name: core.ToolNames.TOOL_CALL, + kind: core.Kind.Other, + description: 'Deferred tool bridge', + build: vi.fn((params: Record) => ({ params })), + }; + const toolSearch = { + name: core.ToolNames.TOOL_SEARCH, + kind: core.Kind.Other, + description: 'Deferred tool discovery', + build: vi.fn((params: Record) => ({ params })), + }; + const target = { + name: 'web_fetch', + kind: core.Kind.Other, + description: 'Fetches a URL', + schema: { + parametersJsonSchema: { + type: 'object', + properties: { + url: { type: 'string' }, + prompt: { type: 'string' }, + }, + required: ['url', 'prompt'], + additionalProperties: false, + }, + }, + build: vi.fn(), + }; + mockToolRegistry.getTool.mockImplementation((name: string) => + name === bridge.name + ? bridge + : name === target.name + ? target + : name === toolSearch.name + ? toolSearch + : undefined, + ); + mockToolRegistry.ensureTool.mockImplementation(async (name: string) => + name === bridge.name + ? bridge + : name === target.name + ? target + : name === toolSearch.name + ? toolSearch + : undefined, + ); + mockToolRegistry.isDeferredAndHidden.mockImplementation( + (name: string) => name === target.name, + ); + const toolLoopState = { + totalToolCalls: 0, + invalidToolParamErrors: new Map(), + toolCallKeyCounts: new Map(), + maxToolCallKeyRepeat: 0, + loopDetected: false, + }; + + const result = await ( + session as unknown as { + runToolCalls: ( + abortSignal: AbortSignal, + promptId: string, + calls: FunctionCall[], + loopState: typeof toolLoopState, + ) => Promise<{ parts: Part[] }>; + } + ).runToolCalls( + new AbortController().signal, + 'prompt-tool-call-bridge-denied', + [ + { + id: 'bridge-denied-call', + name: core.ToolNames.TOOL_CALL, + args: { name: target.name, arguments: {} }, + }, + ], + toolLoopState, + ); + + const errorText = String( + result.parts[0]?.functionResponse?.response?.['error'], + ); + expect( + errorText.startsWith(core.DEFERRED_TOOL_CALL_REFUSAL_PREFIX), + ).toBe(true); + expect(errorText).toContain( + "not permitted by this agent's tool policy", + ); + expect(errorText).not.toContain("required property 'url'"); + expect(toolLoopState.invalidToolParamErrors.size).toBe(0); + expect(target.build).not.toHaveBeenCalled(); + }); + it('marks a disabled ACP tool_call as a bridge refusal', async () => { mockConfig.getPermissionManager = vi.fn().mockReturnValue({ isToolEnabled: vi.fn().mockResolvedValue(false), diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index 7b0affe8692..e0165ef74a8 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -1644,6 +1644,71 @@ describe('CoreToolScheduler', () => { } }); + it('denies a permission-manager-blocked bridged target with the loop denial, not a parameter pre-check refusal', async () => { + // The scheduler applies two execution policies to a bridged target: the + // owner allowlist (forwarded into resolution, tested above) and the + // permission-manager enablement gate in _schedule. A pm-denied target + // with schema-invalid arguments must still surface the loop's own + // denial — with its deny-rule attribution — rather than the bridge's + // INVALID_TOOL_PARAMS pre-check refusal, which would tell the model to + // fix arguments on a call that could never run. Mutation check: + // dropping the suppressArgumentPreCheck wiring in + // resolveToolCallBridgeRequest turns this red with "params must have + // required property 'url'". + const execute = vi.fn(); + const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); + const deferred = new MockTool({ + name: 'web_fetch', + shouldDefer: true, + params: { + type: 'object', + properties: { + url: { type: 'string' }, + prompt: { type: 'string' }, + }, + required: ['url', 'prompt'], + additionalProperties: false, + }, + execute, + }); + const { scheduler, onAllToolCallsComplete } = + createSchedulerForLegacyToolTests({ + toolsByName: new Map([ + [bridge.name, bridge], + [deferred.name, deferred], + ]), + deferredHiddenNames: new Set([deferred.name]), + permissionManager: { + isToolEnabled: async (name: string) => name !== 'web_fetch', + findMatchingDenyRule: () => 'permissions.deny: web_fetch', + }, + }); + + await scheduler.schedule( + { + callId: 'bridge-pm-deny-target', + name: ToolNames.TOOL_CALL, + args: { name: deferred.name, arguments: {} }, + isClientInitiated: false, + prompt_id: 'prompt-bridge-pm-deny-target', + }, + new AbortController().signal, + ); + + expect(execute).not.toHaveBeenCalled(); + const completed = onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.errorType).toBe(ToolErrorType.EXECUTION_DENIED); + expect(completed.response.error?.message).toContain( + 'permissions.deny: web_fetch', + ); + expect(completed.response.error?.message).not.toContain( + "required property 'url'", + ); + } + }); + it('accrues bridge argument refusals per target for retry-loop detection', async () => { // The refusal carries the validated targetName for exactly this // accounting: alternating broken bridged calls against two distinct @@ -1736,6 +1801,192 @@ describe('CoreToolScheduler', () => { } }); + it('does not preserve stale bridge-refusal counters across a policy-denied bridge batch', async () => { + // An EXECUTION_DENIED bridge refusal accrues nothing, so it must not + // keep the denied target's stale retry counters alive through the + // batch-start prune — otherwise the next bridged malformed call fires + // RETRY LOOP DETECTED one failure early. Mutation check: keying the + // presence-set widening on targetName presence alone (instead of the + // INVALID_TOOL_PARAMS error type) turns this red. + const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); + const writeFile = new MockTool({ + name: 'write_file', + shouldDefer: true, + params: { + type: 'object', + properties: { + file_path: { type: 'string' }, + content: { type: 'string' }, + }, + required: ['file_path', 'content'], + additionalProperties: false, + }, + }); + let policyDenies = false; + const { scheduler, onAllToolCallsComplete } = + createSchedulerForLegacyToolTests({ + toolsByName: new Map([ + [bridge.name, bridge], + [writeFile.name, writeFile], + ]), + deferredHiddenNames: new Set([writeFile.name]), + isToolExecutionAllowed: () => !policyDenies, + }); + + const runBatch = async ( + batchId: string, + requests: Array<{ + name: string; + args: Record; + }>, + ) => { + onAllToolCallsComplete.mockClear(); + await scheduler.schedule( + requests.map((request, index) => ({ + callId: `${batchId}-${index}`, + ...request, + isClientInitiated: false, + prompt_id: 'prompt-bridge-stale-counter', + })), + new AbortController().signal, + ); + await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); + return onAllToolCallsComplete.mock.calls[0][0] as ToolCall[]; + }; + + // Seed the target's bridge-channel counter to threshold - 1. + for (const batch of ['seed-1', 'seed-2']) { + const [completed] = await runBatch(batch, [ + { + name: ToolNames.TOOL_CALL, + args: { name: 'write_file', arguments: {} }, + }, + ]); + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.errorType).toBe( + ToolErrorType.INVALID_TOOL_PARAMS, + ); + expect(completed.response.error?.message).not.toContain( + 'RETRY LOOP DETECTED', + ); + } + } + + // A batch whose only request is a policy-denied bridged write_file: + // records nothing, and must not retain the stale counter. + policyDenies = true; + const [denied] = await runBatch('denied', [ + { + name: ToolNames.TOOL_CALL, + args: { name: 'write_file', arguments: {} }, + }, + ]); + expect(denied.status).toBe('error'); + if (denied.status === 'error') { + expect(denied.response.errorType).toBe(ToolErrorType.EXECUTION_DENIED); + } + policyDenies = false; + + // The next bridged malformed call restarts at 1, not threshold. + const [after] = await runBatch('after', [ + { + name: ToolNames.TOOL_CALL, + args: { name: 'write_file', arguments: {} }, + }, + ]); + expect(after.status).toBe('error'); + if (after.status === 'error') { + expect(after.response.errorType).toBe(ToolErrorType.INVALID_TOOL_PARAMS); + expect(after.response.error?.message).not.toContain( + 'RETRY LOOP DETECTED', + ); + } + }); + + it('accrues bridged pre-check refusals independently from the same target’s direct failures', async () => { + // A bridged pre-check refusal and a direct validation failure of the + // SAME target must not share a retry namespace: recordRetryableToolError + // prunes same-prefix keys on every record, so an unmarked shared key + // would let the two channels reset each other every batch and the mixed + // loop would never reach the threshold — the loop this PR exists to + // stop. Mutation check: keying the refusal branch on the bare + // targetName (no channel marker) turns this red. + const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); + const writeFile = new MockTool({ + name: 'write_file', + shouldDefer: true, + params: { + type: 'object', + properties: { + file_path: { type: 'string' }, + content: { type: 'string' }, + }, + required: ['file_path', 'content'], + additionalProperties: false, + }, + }); + const { scheduler, onAllToolCallsComplete } = + createSchedulerForLegacyToolTests({ + toolsByName: new Map([ + [bridge.name, bridge], + [writeFile.name, writeFile], + ]), + deferredHiddenNames: new Set([writeFile.name]), + }); + + const runMixedBatch = async (batchId: number) => { + onAllToolCallsComplete.mockClear(); + await scheduler.schedule( + [ + { + callId: `mixed-${batchId}-bridge`, + name: ToolNames.TOOL_CALL, + args: { name: 'write_file', arguments: {} }, + isClientInitiated: false, + prompt_id: 'prompt-bridge-mixed-channel', + }, + { + callId: `mixed-${batchId}-direct`, + name: 'write_file', + args: {}, + isClientInitiated: false, + prompt_id: 'prompt-bridge-mixed-channel', + }, + ], + new AbortController().signal, + ); + await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); + return onAllToolCallsComplete.mock.calls[0][0] as ToolCall[]; + }; + + for (const batch of [await runMixedBatch(1), await runMixedBatch(2)]) { + expect(batch).toHaveLength(2); + for (const completed of batch) { + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.errorType).toBe( + ToolErrorType.INVALID_TOOL_PARAMS, + ); + expect(completed.response.error?.message).not.toContain( + 'RETRY LOOP DETECTED', + ); + } + } + } + + const third = await runMixedBatch(3); + expect(third).toHaveLength(2); + const [bridgeRefusal, directFailure] = third; + expect(bridgeRefusal.status).toBe('error'); + expect(directFailure.status).toBe('error'); + if (directFailure.status === 'error') { + expect(directFailure.response.error?.message).toContain( + 'RETRY LOOP DETECTED', + ); + } + }); + it('applies the retry-loop directive to repeated invalid tool_call envelopes', async () => { const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); const { scheduler, onAllToolCallsComplete } = @@ -1787,12 +2038,13 @@ describe('CoreToolScheduler', () => { }); it('prunes the bridge-keyed retry counter across a successful bridged execution', async () => { - // R1-18: invalid envelopes record under the model-facing name - // (`tool_call:`), while a successfully resolved envelope renames the - // request to the resolved TARGET before the batch-start prune runs — so - // the prune is the only mechanism that clears a stale `tool_call:` count - // across a successful bridged execution. Interleave one: without the - // prune (e.g. a refactor keying presence by model-facing name), the count + // R1-18: invalid envelopes record under the bridge channel + // (`tool_call(via tool_call):`), while a successfully resolved + // envelope renames the request to the resolved TARGET before the + // batch-start prune runs — so the prune is the only mechanism that + // clears a stale bridge-channel count across a successful bridged + // execution. Interleave one: without the prune (e.g. a refactor keying + // presence by model-facing name), the count // of 2 would survive the successful call and the next two identical // failures would reach the threshold and inject RETRY LOOP DETECTED // prematurely — while the direct-tool isolation test stays green, because diff --git a/packages/core/src/core/coreToolScheduler.ts b/packages/core/src/core/coreToolScheduler.ts index fdb7b993183..e0208a70a44 100644 --- a/packages/core/src/core/coreToolScheduler.ts +++ b/packages/core/src/core/coreToolScheduler.ts @@ -1088,6 +1088,20 @@ function getModelFacingToolName(request: ToolCallRequestInfo): string { ); } +/** + * Retry-accounting namespace for bridge pre-check refusals. The refusal fires + * before the request is rewritten to the target, but recordRetryableToolError + * prunes same-prefix keys — so keying it on the bare target name would let a + * bridged refusal and a direct validation failure of the SAME target reset + * each other every batch and neither channel would ever reach + * VALIDATION_RETRY_LOOP_THRESHOLD. The suffix keeps the two channels + * independent while the per-batch presence prune still recognizes the entry + * as belonging to the target. + */ +function bridgeRetryToolName(targetName: string): string { + return `${targetName}(via tool_call)`; +} + // NOTE: the `⚠` in this and TRUNCATION_RETRY_LOOP_DIRECTIVE below is part of an // LLM-facing prompt directive (injected into the model prompt, not rendered in // the TUI). The width-1 glyph rationale used elsewhere in this change does not @@ -2722,6 +2736,21 @@ export class CoreToolScheduler { // The owner policy must win over the argument pre-check so a denied // target keeps its specific EXECUTION_DENIED. isTargetExecutionAllowed: this.isToolExecutionAllowed, + // The permission-manager gate in _schedule owns the richer denial + // (deny-rule attribution); a pm-denied target skips the argument + // pre-check so that denial — not a parameter error for a call that + // could never run — is what the model sees. + suppressArgumentPreCheck: permissionManager + ? async (targetName: string) => { + try { + return !(await permissionManager.isToolEnabled(targetName)); + } catch { + // A policy lookup failure must not swallow the argument + // pre-check; the loop's permission gate reports the error. + return false; + } + } + : undefined, }), ); if ('error' in resolution) { @@ -2735,22 +2764,10 @@ export class CoreToolScheduler { }; } - // The pre-schedule gates saw the wrapper, so apply the owner's policy to - // the resolved target before execution. - if ( - this.isToolExecutionAllowed && - !this.isToolExecutionAllowed(resolution.tool.name) - ) { - return { - ...request, - bridgeResolutionError: { - error: new Error( - `Tool "${resolution.tool.name}" is not permitted by this agent's tool policy (execution allowlist or disallowedTools blocklist).`, - ), - type: ToolErrorType.EXECUTION_DENIED, - }, - }; - } + // No post-resolution isToolExecutionAllowed gate here: resolution + // already consulted the same predicate (constructor-fixed) on the same + // target name, so a second copy would be unreachable and free to + // diverge. return { ...request, @@ -2910,13 +2927,21 @@ export class CoreToolScheduler { // whenever any current request matched caused stale counts for // unrelated tools to survive and fire RETRY LOOP DETECTED prematurely // the next time those tools were used. A refused bridge request keeps - // the wrapper name (`tool_call`), so its validated target name must - // join the presence set alongside it. + // the wrapper name (`tool_call`), so the channel-marked name of the + // validated target must join the presence set alongside it — but only + // for INVALID_TOOL_PARAMS refusals, the one error type that accrues + // below: an EXECUTION_DENIED (policy) refusal records nothing, so it + // must not keep the denied target's stale counters alive either. if (this.validationRetryCounts.size > 0) { const currentToolNames = new Set( requestsToProcess.flatMap((r) => - r.bridgeResolutionError?.targetName !== undefined - ? [r.name, r.bridgeResolutionError.targetName] + r.bridgeResolutionError?.type === + ToolErrorType.INVALID_TOOL_PARAMS && + r.bridgeResolutionError.targetName !== undefined + ? [ + r.name, + bridgeRetryToolName(r.bridgeResolutionError.targetName), + ] : [r.name], ), ); @@ -3001,9 +3026,16 @@ export class CoreToolScheduler { ) { // Key on the validated target, not the `tool_call` wrapper: // alternating failures against distinct targets must accrue - // per-target instead of pruning each other's counter. + // per-target instead of pruning each other's counter. The + // channel marker keeps a target's bridged pre-check refusals + // from prefix-pruning (or being pruned by) the SAME target's + // direct validation failures — see bridgeRetryToolName. const count = recordBatchRetryableToolError( - reqInfo.bridgeResolutionError.targetName ?? reqInfo.name, + reqInfo.bridgeResolutionError.targetName !== undefined + ? bridgeRetryToolName( + reqInfo.bridgeResolutionError.targetName, + ) + : reqInfo.name, bridgeError.message, ); if (count >= VALIDATION_RETRY_LOOP_THRESHOLD) { diff --git a/packages/core/src/tools/agent/agent.test.ts b/packages/core/src/tools/agent/agent.test.ts index 5c450ca819e..aa3a70dd937 100644 --- a/packages/core/src/tools/agent/agent.test.ts +++ b/packages/core/src/tools/agent/agent.test.ts @@ -1456,12 +1456,14 @@ describe('AgentTool', () => { expect(result).toBeNull(); }); - it('coalesces a second synchronous refresh kick onto the in-flight refresh', async () => { - // An unknown subagent_type kicks refreshSubagents so the cache catches - // up — but validation can fire twice for one call (the tool_call - // bridge pre-check validates ahead of build(), #12889), and each kick - // used to start its own full subagent rescan + llmClient.setTools. A - // kick arriving while a refresh is in flight must coalesce onto it. + it('coalesces a synchronous refresh kick onto the in-flight refresh and arms one follow-up', async () => { + // An unknown subagent_type kicks a refresh so the cache catches up — + // the kick is the ONLY discovery path for agent files created out of + // band (the change listener does not fire for those), and the + // in-flight scan may have read the directory before the file existed, + // so a kick landing mid-refresh must queue one follow-up scan instead + // of being dropped. The second synchronous kick still coalesces onto + // the in-flight scan (one scan, one arm — not three scans). await agentTool.refreshSubagents(); const listSpy = vi.mocked(mockSubagentManager.listSubagents); listSpy.mockClear(); @@ -1476,13 +1478,18 @@ describe('AgentTool', () => { }); expect(listSpy).toHaveBeenCalledTimes(1); - // After the in-flight refresh settles, a later kick re-scans. + // The armed follow-up re-scans once the in-flight refresh settles — + // no third kick needed. Mutation check: kicking plain + // refreshSubagents() (dropping the arm) leaves this at 1. await vi.runAllTimersAsync(); + expect(listSpy).toHaveBeenCalledTimes(2); + + // After everything settled, a later kick re-scans again. agentTool.validateToolParams({ ...validParams, subagent_type: 'missing', }); - expect(listSpy).toHaveBeenCalledTimes(2); + expect(listSpy).toHaveBeenCalledTimes(3); }); it('should reject empty description', async () => { @@ -2858,6 +2865,111 @@ describe('AgentTool', () => { expect(agentTool.description).toContain('A brand new agent'); }); + it('arms exactly one follow-up scan when a change lands mid-refresh', async () => { + // A change landing mid-refresh is not reflected in the in-flight scan + // (it may have read the directory before the file existed), so the + // listener arms one follow-up instead of coalescing into the stale + // read — and two change events during one refresh must yield ONE + // follow-up, not two. Mutation checks: coalescing listener kicks like + // validation kicks (bare return when a refresh is in flight) leaves + // the count at 1; dropping the arm-once guard yields 3. + const listSpy = vi.mocked(mockSubagentManager.listSubagents); + listSpy.mockClear(); + let releaseScan!: () => void; + listSpy.mockImplementationOnce( + () => + new Promise((resolve) => { + releaseScan = () => resolve(mockSubagents); + }), + ); + + void agentTool.refreshSubagents(); + expect(listSpy).toHaveBeenCalledTimes(1); + + const listener = changeListeners[0]; + listener?.(); + listener?.(); + // Both change events coalesced into one armed follow-up; the + // in-flight scan is still the only scan so far. + expect(listSpy).toHaveBeenCalledTimes(1); + + releaseScan(); + await vi.runAllTimersAsync(); + expect(listSpy).toHaveBeenCalledTimes(2); + }); + + it('does not run an armed follow-up refresh after dispose', async () => { + // dispose() removes the change listener, but an already-armed + // follow-up is a separate promise chain: without a disposal guard it + // would still run a full rescan + llmClient.setTools() on a torn-down + // tool. Mutation check: dropping the disposed check from the armed + // callback turns this red (a second scan runs). + const listSpy = vi.mocked(mockSubagentManager.listSubagents); + listSpy.mockClear(); + let releaseScan!: () => void; + listSpy.mockImplementationOnce( + () => + new Promise((resolve) => { + releaseScan = () => resolve(mockSubagents); + }), + ); + + void agentTool.refreshSubagents(); + expect(listSpy).toHaveBeenCalledTimes(1); + + changeListeners[0]?.(); + agentTool.dispose(); + + releaseScan(); + await vi.runAllTimersAsync(); + expect(listSpy).toHaveBeenCalledTimes(1); + }); + + it('absorbs a setTools rejection from an armed follow-up refresh', async () => { + // runRefreshSubagents can reject (its finally awaits + // llmClient.setTools()), and .finally() propagates that rejection into + // the armed follow-up chain — without a .catch at the void boundary it + // floats as an unhandledRejection, which this file's own contract + // forbids. The follow-up scan must still run. Mutation check: dropping + // the armed chain's .catch(...) turns the rejection assertion red. + const setTools = vi.fn().mockRejectedValue(new Error('setTools failed')); + vi.mocked(config.getLlmClient).mockReturnValue({ + setTools, + } as unknown as ReturnType); + const unhandledRejections: unknown[] = []; + const onUnhandledRejection = (reason: unknown) => { + unhandledRejections.push(reason); + }; + process.on('unhandledRejection', onUnhandledRejection); + try { + const listSpy = vi.mocked(mockSubagentManager.listSubagents); + listSpy.mockClear(); + let releaseScan!: () => void; + listSpy.mockImplementationOnce( + () => + new Promise((resolve) => { + releaseScan = () => resolve(mockSubagents); + }), + ); + + // The in-flight refresh rejects in its finally (setTools); the test + // swallows that one directly so only the armed chain is measured. + const inFlight = agentTool.refreshSubagents(); + void inFlight.catch(() => {}); + changeListeners[0]?.(); + + releaseScan(); + await vi.runAllTimersAsync(); + + // The armed follow-up still ran (and hit the rejecting setTools + // again), and no rejection escaped either void boundary. + expect(listSpy).toHaveBeenCalledTimes(2); + expect(unhandledRejections).toEqual([]); + } finally { + process.removeListener('unhandledRejection', onUnhandledRejection); + } + }); + it('should refresh available subagents and update description', async () => { const newSubagents: SubagentConfig[] = [ { diff --git a/packages/core/src/tools/agent/agent.ts b/packages/core/src/tools/agent/agent.ts index e63729ae229..e5183a488be 100644 --- a/packages/core/src/tools/agent/agent.ts +++ b/packages/core/src/tools/agent/agent.ts @@ -886,7 +886,7 @@ export class AgentTool extends BaseDeclarativeTool { this.delegationSurface = resolveAgentDelegationSurface(config); this.subagentManager = config.getSubagentManager(); this.removeChangeListener = this.subagentManager.addChangeListener(() => { - this.refreshSubagentsFromListener(); + this.requestRefresh(); }); // Initialize the tool asynchronously @@ -894,20 +894,23 @@ export class AgentTool extends BaseDeclarativeTool { } dispose(): void { + this.disposed = true; this.removeChangeListener(); } private refreshInFlight: Promise | undefined; private listenerRefreshArmed = false; + private disposed = false; /** * Asynchronously initializes the tool by loading available subagents * and updating the description and schema. * - * Concurrent kicks coalesce onto the in-flight refresh: validateToolParams - * fires one per validated call (and the bridge argument pre-check doubles - * that for a bridged Agent call), and a second scan + setTools mid-turn - * buys nothing — the in-flight scan already reads the current state. + * Concurrent callers coalesce onto the in-flight refresh: a second scan + + * setTools mid-turn buys nothing on its own. Signals that mean "subagent + * state changed" (the change listener, the unknown-subagent_type + * validation kick) go through requestRefresh() instead, because the + * in-flight scan may have read the directory before the change landed. */ refreshSubagents(): Promise { this.refreshInFlight ??= this.runRefreshSubagents().finally(() => { @@ -917,22 +920,35 @@ export class AgentTool extends BaseDeclarativeTool { } /** - * The change listener must not be dropped like a validation kick: a change - * landing mid-refresh is not reflected in the in-flight scan, so arm one - * follow-up refresh instead of coalescing into the stale read. + * Fire-and-forget refresh for out-of-band signals (the change listener and + * the validation kick — the kick is the only discovery path for agent + * files created outside the manager, which the listener never sees). A + * signal landing mid-refresh is not reflected in the in-flight scan, so + * arm exactly one follow-up refresh instead of coalescing into the stale + * read. The voided chains are .catch()-guarded per this file's contract: + * runRefreshSubagents can reject (its finally awaits llmClient.setTools()). */ - private refreshSubagentsFromListener(): void { + private requestRefresh(): void { + if (this.disposed) { + return; + } if (this.refreshInFlight) { if (!this.listenerRefreshArmed) { this.listenerRefreshArmed = true; - void this.refreshInFlight.finally(() => { - this.listenerRefreshArmed = false; - void this.refreshSubagents(); - }); + void this.refreshInFlight + .finally(() => { + this.listenerRefreshArmed = false; + this.requestRefresh(); + }) + .catch((error) => + debugLogger.warn('Follow-up subagent refresh failed:', error), + ); } return; } - void this.refreshSubagents(); + void this.refreshSubagents().catch((error) => + debugLogger.warn('Subagent refresh failed:', error), + ); } private async runRefreshSubagents(): Promise { @@ -1161,8 +1177,11 @@ The background-agent rules above apply to background forks unchanged.${delegatio // resolves the type via loadSubagent(), which reads from disk and // fails with a clear "not found" error if the agent truly doesn't // exist. Kick a refresh (validation must stay synchronous) so the - // cache and schema catch up for subsequent calls. - void this.refreshSubagents(); + // cache and schema catch up for subsequent calls. The kick goes + // through requestRefresh(): a kick landing mid-scan arms one + // follow-up, because the in-flight scan may have read the + // directory before the file existed. + this.requestRefresh(); } } } diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 96a0e3a3d9c..d8aeb0c0442 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -4,7 +4,7 @@ * SPDX-License-Identifier: Apache-2.0 */ -import { describe, expect, it } from 'vitest'; +import { describe, expect, it, vi } from 'vitest'; import { MockTool } from '../test-utils/mock-tool.js'; import { runWithAgentContext } from '../agents/runtime/agent-context.js'; import { runWithTeammateIdentity } from '../agents/team/identity.js'; @@ -425,6 +425,35 @@ describe('ToolCallTool', () => { }); }); + it('denies a policy-blocked target ahead of the hidden-tool gate', async () => { + // Registered NOT hidden: the caller's execution-policy gate must fire + // BEFORE the isDeferredAndHidden check, like the sibling plan-lifecycle + // and leader-only gates above. Otherwise a policy-disabled but visible + // target bridged through tool_call gets the "already visible — call it + // directly" INVALID_TOOL_PARAMS, telling the model to call a tool the + // owner's policy forbids (and on ACP accruing an invalid-parameter + // strike). Mutation check: moving the isTargetExecutionAllowed gate + // below the isDeferredAndHidden check turns this red with the + // "already visible" refusal; dropping targetName from the denial also + // turns it red. + const target = new MockTool({ name: 'web_fetch', shouldDefer: false }); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set()), + { name: target.name, arguments: {} }, + { isTargetExecutionAllowed: async () => false }, + ); + + expect(result).toMatchObject({ + errorType: ToolErrorType.EXECUTION_DENIED, + targetName: 'web_fetch', + error: expect.objectContaining({ + message: expect.stringContaining( + "not permitted by this agent's tool policy", + ), + }), + }); + }); + it.each([ ToolNames.TEAM_DELETE, ToolNames.WORKFLOW, @@ -809,6 +838,43 @@ describe('ToolCallTool', () => { expect(result).toMatchObject({ arguments: { count: '3' } }); }); + it('pre-checks only the schema layer, leaving value-level rules to build()', async () => { + // The pre-check exists to name the target and the missing field in + // the refusal (#12889); a target's value-level rules (fs stats, + // content scans, the AgentTool refresh kick) must run exactly once, + // at build() time — running them here would pay their side effects + // twice per bridged call. Mutation check: routing the pre-check + // through target.validateToolParams (schema + value rules) fires the + // spy and turns this red. + const valueRuleSpy = vi.fn( + (_params: { [key: string]: unknown }): string | null => null, + ); + class ValueRuleTool extends MockTool { + protected override validateToolParamValues(params: { + [key: string]: unknown; + }): string | null { + return valueRuleSpy(params); + } + } + const target = new ValueRuleTool({ + name: 'write_file', + shouldDefer: true, + params: { + type: 'object', + properties: { file_path: { type: 'string' } }, + required: ['file_path'], + }, + }); + + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: 'write_file', arguments: { file_path: '/tmp/a.txt' } }, + ); + + expect(result).not.toHaveProperty('error'); + expect(valueRuleSpy).not.toHaveBeenCalled(); + }); + // The shared native shape of the omni media-policy family: io params // with `resourceId` as the model-facing `inputPath` alternative, and // only `outputDir` required natively (0 of the 14 shipped tools require @@ -968,16 +1034,17 @@ describe('ToolCallTool', () => { } }); - it('resolves a target whose validateToolParams throws, leaving the throw to build()', async () => { - // A throwing validator must not become a new bridge failure mode: the - // scheduler's build() reports the same throw as before. Mutation - // check: dropping the try/catch around the pre-check turns this red. - class ThrowingValidatorTool extends MockTool { - override validateToolParams(): string | null { - throw new Error('boom from validator'); + it('resolves a target whose schema access throws, leaving the throw to build()', async () => { + // A target whose declaration throws under the pre-check must not + // become a new bridge failure mode: the scheduler's build() reports + // the same throw as before. Mutation check: dropping the try/catch + // around the pre-check turns this red. + class ThrowingSchemaTool extends MockTool { + override get schema(): never { + throw new Error('boom from schema access'); } } - const target = new ThrowingValidatorTool({ + const target = new ThrowingSchemaTool({ name: 'throwing_tool', shouldDefer: true, params: { type: 'object', properties: {} }, diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 641368e6300..0e571c14d13 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -68,6 +68,15 @@ export interface DeferredToolCallOptions { * instead of surfacing a parameter error for a call that could never run. */ isTargetExecutionAllowed?: (targetName: string) => boolean | Promise; + /** + * Set when the caller applies its own target policy downstream of + * resolution (the scheduler's permission-manager gate) with a richer + * denial message than the bridge can build. Returning true skips the + * argument pre-check for that target so the caller's own denial — not a + * parameter error for a call that could never run — is what the model + * sees. + */ + suppressArgumentPreCheck?: (targetName: string) => boolean | Promise; } export const DEFERRED_TOOL_CALL_REFUSAL_PREFIX = '[tool_call bridge refused] '; @@ -266,43 +275,40 @@ export async function resolveDeferredToolCall( // The bridge envelope deliberately types `arguments` as a bare object (the // declaration must stay byte-stable across catalog changes), so `{}` is // envelope-valid even when the target requires fields. Pre-validate against - // the target's own schema so the refusal names the target and the missing - // field, instead of surfacing a bare Ajv message after the call has been - // unwrapped (#12889). Validate a clone: SchemaValidator coerces values in - // place, and the scheduler re-validates the returned arguments at build + // the target's model-visible schema so the refusal names the target and the + // missing field, instead of surfacing a bare Ajv message after the call has + // been unwrapped (#12889). Validate a clone: SchemaValidator coerces values + // in place, and the scheduler re-validates the returned arguments at build // time. // - // Omni media-policy targets are pre-checked against the model-visible - // projection (their `schema` getter) instead of the native schema: both - // frontends run the modelAccess gate AFTER bridge resolution - // (coreToolScheduler's `evaluateMediaPolicyToolCall`, before buildInvocation; - // ACP Session.runTool), and that gate resolves `resourceId` → `inputPath` - // and merges `defaultArguments`/`lockedArguments`. Their `validateToolParams` - // deliberately checks the NATIVE schema plus io value rules that assume - // that completion, so running it here refuses calls the very next stage - // accepts — and for an operator locked key the refusal is unwinnable both - // ways (omitting it fails natively, sending it fails the gate). The - // projection has locked keys stripped from `required`, so validating it - // still refuses a missing model-visible required field while never - // demanding a key the model is forbidden to send. `mediaPolicyDescriptor` - // is the code-level fact the gate itself keys off (it passes every - // non-policy tool through untouched), so the narrowed check covers exactly - // the tools whose arguments a downstream stage completes. Nothing fails - // open: the gate still emits named `invalid_params` refusals, and build() - // re-validates the merged arguments against the native schema. - const isMediaPolicyTarget = target.mediaPolicyDescriptor !== undefined; + // Only the schema layer runs here — never the target's full + // validateToolParams: its value-level rules (fs stats, content scans, the + // AgentTool refresh kick) run unchanged at build() time, so running them + // here would pay their side effects twice per bridged call. The + // model-visible `schema` getter is also what makes this safe for omni + // media-policy targets: their declaration is a projection with operator + // `lockedArguments` stripped from `required`, while their + // `validateToolParams` deliberately checks the NATIVE schema plus io value + // rules that assume the modelAccess gate (which both frontends run AFTER + // bridge resolution) has merged those arguments back in — running it here + // would refuse calls the very next stage accepts. let paramsError: string | null = null; // A truncated response yields to the caller's truncation handling: the // arguments are incomplete for transport reasons, not a schema misreading. - if (!options?.wasOutputTruncated) { + // A target the caller's own downstream policy gate will deny (the + // scheduler's permission-manager gate owns the richer denial message) + // must surface that denial, not a parameter error for a call that could + // never run. + const preCheckSuppressed = + options?.suppressArgumentPreCheck !== undefined && + (await options.suppressArgumentPreCheck(target.name)); + if (!options?.wasOutputTruncated && !preCheckSuppressed) { try { const argsClone = structuredClone(invocation.params.arguments); - paramsError = isMediaPolicyTarget - ? SchemaValidator.validate( - target.schema.parametersJsonSchema, - argsClone, - ) - : target.validateToolParams(argsClone); + paramsError = SchemaValidator.validate( + target.schema.parametersJsonSchema, + argsClone, + ); } catch { // A target whose validation throws under this pre-check must not become // a new bridge failure mode: the scheduler's build() reports the same From bad821c2a994b6e5338615e0b4d1400bc997be6a Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Tue, 29 Sep 2026 16:04:05 +0800 Subject: [PATCH 06/27] fix(core): clone the pre-check schema and log the bridge policy lookup failure Two review findings on the bridged tool_call argument pre-validation: - tool-call: validate a per-call structural copy of the target's parametersJsonSchema. Ajv caches a compiled schema by object identity for the life of the process, and AgentTool mutates its own parameterSchema in place, so the first bridged call pinned every later one to the pre-refresh shape and refused a property the target had since advertised. The copy resolves through the JSON-text tier, which still shares one compiled validator per distinct schema text. - coreToolScheduler: log the permission-manager lookup failure that the suppressArgumentPreCheck guard swallows. On the refusal path _schedule continues ahead of the permission gate, so the catch was the only place that saw the error and it recorded nothing. Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-closeout/jmumcpwj7eg --- .../core/src/core/coreToolScheduler.test.ts | 78 +++++++++++++++++++ packages/core/src/core/coreToolScheduler.ts | 12 ++- packages/core/src/tools/tool-call.test.ts | 45 +++++++++++ packages/core/src/tools/tool-call.ts | 13 +++- 4 files changed, 141 insertions(+), 7 deletions(-) diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index e0165ef74a8..97156bd1202 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -1709,6 +1709,84 @@ describe('CoreToolScheduler', () => { } }); + it('logs the policy lookup failure and still pre-checks a bridged target', async () => { + // The try/catch around the permission-manager lookup is the only thing + // keeping a policy-store rejection out of _schedule's Promise.all, whose + // try closes with a bare finally: an uncaught rejection there rejects the + // public schedule() and fails the whole batch instead of one call. The + // guard returns false ("do not suppress"), so the pre-check still runs and + // still names the missing field. The swallow is logged because _schedule + // pushes the bridge refusal and continues ahead of the permission gate, so + // nothing else records that the lookup failed. Mutation check: inlining + // !(await permissionManager.isToolEnabled(...)) without the try turns this + // red — schedule() rejects, or the call completes EXECUTION_DENIED instead + // of INVALID_TOOL_PARAMS. Deleting only the debugLogger.warn turns the log + // assertion red. + const execute = vi.fn(); + const policyError = new Error('policy store unavailable'); + const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); + const deferred = new MockTool({ + name: 'web_fetch', + shouldDefer: true, + params: { + type: 'object', + properties: { + url: { type: 'string' }, + prompt: { type: 'string' }, + }, + required: ['url', 'prompt'], + additionalProperties: false, + }, + execute, + }); + const { scheduler, onAllToolCallsComplete } = + createSchedulerForLegacyToolTests({ + toolsByName: new Map([ + [bridge.name, bridge], + [deferred.name, deferred], + ]), + deferredHiddenNames: new Set([deferred.name]), + permissionManager: { + isToolEnabled: async (name: string) => { + if (name === 'web_fetch') { + throw policyError; + } + return true; + }, + findMatchingDenyRule: () => undefined, + }, + }); + debugLoggerWarnSpy.mockClear(); + + await scheduler.schedule( + { + callId: 'bridge-policy-lookup-throws', + name: ToolNames.TOOL_CALL, + args: { name: deferred.name, arguments: {} }, + isClientInitiated: false, + prompt_id: 'prompt-bridge-policy-lookup-throws', + }, + new AbortController().signal, + ); + + expect(execute).not.toHaveBeenCalled(); + expect(debugLoggerWarnSpy).toHaveBeenCalledWith( + expect.stringContaining('Bridge pre-check policy lookup failed for'), + deferred.name, + policyError, + ); + const completed = onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.errorType).toBe( + ToolErrorType.INVALID_TOOL_PARAMS, + ); + expect(completed.response.error?.message).toContain( + "required property 'url'", + ); + } + }); + it('accrues bridge argument refusals per target for retry-loop detection', async () => { // The refusal carries the validated targetName for exactly this // accounting: alternating broken bridged calls against two distinct diff --git a/packages/core/src/core/coreToolScheduler.ts b/packages/core/src/core/coreToolScheduler.ts index e0208a70a44..674a03b16b1 100644 --- a/packages/core/src/core/coreToolScheduler.ts +++ b/packages/core/src/core/coreToolScheduler.ts @@ -2744,9 +2744,15 @@ export class CoreToolScheduler { ? async (targetName: string) => { try { return !(await permissionManager.isToolEnabled(targetName)); - } catch { - // A policy lookup failure must not swallow the argument - // pre-check; the loop's permission gate reports the error. + } catch (error) { + // Do not let a policy lookup failure swallow the pre-check. + // On the refusal path _schedule continues ahead of the + // permission gate, so this is the lookup's only record. + debugLogger.warn( + 'Bridge pre-check policy lookup failed for', + targetName, + error, + ); return false; } } diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index d8aeb0c0442..76db2af3be4 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -838,6 +838,51 @@ describe('ToolCallTool', () => { expect(result).toMatchObject({ arguments: { count: '3' } }); }); + it('re-reads a target schema the target mutates in place after the first call', async () => { + // AgentTool's refresh mutates its own parameterSchema object in place + // (it adds and removes `model`/`name`), and Ajv caches a compiled + // schema by object identity for the life of the process. Handing the + // validator that same object pins every later bridged call to the + // shape the first one happened to compile, so a property the target + // advertises after that first call is refused forever. + // Mutation check: passing target.schema.parametersJsonSchema by + // reference instead of a clone turns this red with "must NOT have + // additional properties". + const target = new MockTool({ + name: 'agent_like', + shouldDefer: true, + params: { + type: 'object', + properties: { prompt: { type: 'string' } }, + required: ['prompt'], + additionalProperties: false, + }, + }); + const registry = makeRegistry([target], new Set([target.name])); + + // The first bridged call compiles the pre-refresh schema. + await resolveDeferredToolCall(registry, { + name: 'agent_like', + arguments: { prompt: 'do the thing' }, + }); + + // The refresh then advertises a new property on that SAME object. + const schema = target.schema.parametersJsonSchema as { + properties: Record; + }; + schema.properties.model = { type: 'string', enum: ['fast', 'pro'] }; + + const result = await resolveDeferredToolCall(registry, { + name: 'agent_like', + arguments: { prompt: 'do the thing', model: 'fast' }, + }); + + expect(result).not.toHaveProperty('error'); + expect(result).toMatchObject({ + arguments: { prompt: 'do the thing', model: 'fast' }, + }); + }); + it('pre-checks only the schema layer, leaving value-level rules to build()', async () => { // The pre-check exists to name the target and the missing field in // the refusal (#12889); a target's value-level rules (fs stats, diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 0e571c14d13..fd014a8050c 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -277,9 +277,14 @@ export async function resolveDeferredToolCall( // envelope-valid even when the target requires fields. Pre-validate against // the target's model-visible schema so the refusal names the target and the // missing field, instead of surfacing a bare Ajv message after the call has - // been unwrapped (#12889). Validate a clone: SchemaValidator coerces values - // in place, and the scheduler re-validates the returned arguments at build - // time. + // been unwrapped (#12889). Validate clones of both sides: SchemaValidator + // coerces argument values in place and the scheduler re-validates them at + // build time; and Ajv caches a compiled schema by object identity, so a + // target that mutates its own schema object in place (AgentTool's refresh + // adds and removes `model`/`name`) would otherwise stay pinned to whatever + // shape it had on the first bridged call. A per-call copy resolves through + // the JSON-text tier, which still shares one compiled validator per + // distinct schema text. // // Only the schema layer runs here — never the target's full // validateToolParams: its value-level rules (fs stats, content scans, the @@ -306,7 +311,7 @@ export async function resolveDeferredToolCall( try { const argsClone = structuredClone(invocation.params.arguments); paramsError = SchemaValidator.validate( - target.schema.parametersJsonSchema, + structuredClone(target.schema.parametersJsonSchema), argsClone, ); } catch { From 888727d28ac915b00a0510dd899cbc2303ef77af Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Tue, 29 Sep 2026 16:11:40 +0800 Subject: [PATCH 07/27] test(core): use index-signature access for the mutated schema property TS4111 rejects dot access on a Record index signature. Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-closeout/jmumcpwj7eg --- packages/core/src/tools/tool-call.test.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 76db2af3be4..4bc500f3e8d 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -870,7 +870,7 @@ describe('ToolCallTool', () => { const schema = target.schema.parametersJsonSchema as { properties: Record; }; - schema.properties.model = { type: 'string', enum: ['fast', 'pro'] }; + schema.properties['model'] = { type: 'string', enum: ['fast', 'pro'] }; const result = await resolveDeferredToolCall(registry, { name: 'agent_like', From 0fd2661a54bf52cf9c58ac236cab1c17eafc9b2a Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Tue, 29 Sep 2026 21:56:33 +0800 Subject: [PATCH 08/27] fix(core): report an in-flight subagent refresh rejection under its own label The armed follow-up chain attached its .catch AFTER .finally. Because .finally propagates the original rejection and its callback is synchronous, that handler could only ever receive the IN-FLIGHT refresh's rejection while its message said 'Follow-up subagent refresh failed'. runRefreshSubagents' finally awaits llmClient.setTools(), which can reject, so a transient setTools failure was reported as a failed follow-up scan even when the follow-up succeeded. The follow-up's own rejection is already reported by requestRefresh's 'Subagent refresh failed' handler. Move the .catch before the .finally so each rejection keeps its own label; the rejection stays handled (no unhandledRejection) and the arm still runs. Tests: extend 'absorbs a setTools rejection from an armed follow-up refresh' to assert both labels, and add 'reports an in-flight refresh rejection under the in-flight label only' (setTools rejects once, then resolves) pinning that the follow-up label is never logged for the in-flight error. Both go red if the chain is reordered back. Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-closeout/jmumpkv7rf3 --- packages/core/src/tools/agent/agent.test.ts | 100 +++++++++++++++++++- packages/core/src/tools/agent/agent.ts | 14 ++- 2 files changed, 106 insertions(+), 8 deletions(-) diff --git a/packages/core/src/tools/agent/agent.test.ts b/packages/core/src/tools/agent/agent.test.ts index aa3a70dd937..d246052dd1b 100644 --- a/packages/core/src/tools/agent/agent.test.ts +++ b/packages/core/src/tools/agent/agent.test.ts @@ -91,6 +91,29 @@ function escapeRegExp(value: string): string { vi.mock('../../subagents/subagent-manager.js'); vi.mock('../../agents/runtime/agent-headless.js'); +// Records the AGENT-tagged debugLogger.warn() calls so tests can assert which +// refresh-failure label a rejection was reported under. Wraps the real logger +// instead of replacing it, so nothing else in this file changes behaviour — +// same shape as goals/goal-persistence.test.ts. +const agentWarnCalls = vi.hoisted(() => [] as unknown[][]); +vi.mock('../../utils/debugLogger.js', async (importOriginal) => { + const original = + await importOriginal(); + return { + ...original, + createDebugLogger: (tag?: string) => { + const logger = original.createDebugLogger(tag); + return { + ...logger, + warn: (...args: unknown[]) => { + if (tag === 'AGENT') agentWarnCalls.push(args); + logger.warn(...args); + }, + }; + }, + }; +}); + // Spies for the subagent-span layer so tests can assert what status taxonomy // was published. The real runInSubagentSpanContext sets up OTel context-with, // which is irrelevant here — we just need the body to run. Review wenshao @@ -2928,10 +2951,13 @@ describe('AgentTool', () => { it('absorbs a setTools rejection from an armed follow-up refresh', async () => { // runRefreshSubagents can reject (its finally awaits // llmClient.setTools()), and .finally() propagates that rejection into - // the armed follow-up chain — without a .catch at the void boundary it - // floats as an unhandledRejection, which this file's own contract - // forbids. The follow-up scan must still run. Mutation check: dropping - // the armed chain's .catch(...) turns the rejection assertion red. + // the armed chain — without a .catch at the void boundary it floats as + // an unhandledRejection, which this file's own contract forbids. The + // follow-up scan must still run. Mutation checks: dropping the armed + // chain's .catch(...) turns the rejection assertion red, and moving it + // back AFTER the .finally() turns the label assertions red (the + // in-flight rejection would be reported as a follow-up failure). + agentWarnCalls.length = 0; const setTools = vi.fn().mockRejectedValue(new Error('setTools failed')); vi.mocked(config.getLlmClient).mockReturnValue({ setTools, @@ -2965,6 +2991,72 @@ describe('AgentTool', () => { // again), and no rejection escaped either void boundary. expect(listSpy).toHaveBeenCalledTimes(2); expect(unhandledRejections).toEqual([]); + // Each rejection is reported under its own label: the in-flight one by + // the armed chain's .catch, the follow-up's own by requestRefresh's. + const labels = agentWarnCalls.map((args) => args[0]); + expect(labels).toContain('In-flight subagent refresh failed:'); + expect(labels).toContain('Subagent refresh failed:'); + expect(labels).not.toContain('Follow-up subagent refresh failed:'); + } finally { + process.removeListener('unhandledRejection', onUnhandledRejection); + } + }); + + it('reports an in-flight refresh rejection under the in-flight label only', async () => { + // The label half of the armed chain: .finally() propagates the ORIGINAL + // rejection and its callback is synchronous, so a .catch placed after + // it can only ever see the in-flight refresh's error — and would name + // it "Follow-up ... failed" even when the follow-up scan succeeded. + // Here setTools rejects once (the in-flight refresh) and then resolves + // (the follow-up). Mutation check: reordering the armed chain back to + // .finally(...).catch('Follow-up subagent refresh failed:') turns the + // in-flight label assertion red. + agentWarnCalls.length = 0; + const setTools = vi + .fn() + .mockRejectedValueOnce(new Error('IN-FLIGHT-setTools-rejected')) + .mockResolvedValue(undefined); + vi.mocked(config.getLlmClient).mockReturnValue({ + setTools, + } as unknown as ReturnType); + const unhandledRejections: unknown[] = []; + const onUnhandledRejection = (reason: unknown) => { + unhandledRejections.push(reason); + }; + process.on('unhandledRejection', onUnhandledRejection); + try { + const listSpy = vi.mocked(mockSubagentManager.listSubagents); + listSpy.mockClear(); + let releaseScan!: () => void; + listSpy.mockImplementationOnce( + () => + new Promise((resolve) => { + releaseScan = () => resolve(mockSubagents); + }), + ); + + const inFlight = agentTool.refreshSubagents(); + void inFlight.catch(() => {}); + changeListeners[0]?.(); + + releaseScan(); + await vi.runAllTimersAsync(); + + // The follow-up scan ran and its setTools succeeded. + expect(listSpy).toHaveBeenCalledTimes(2); + expect(setTools).toHaveBeenCalledTimes(2); + expect(unhandledRejections).toEqual([]); + + const labels = agentWarnCalls.map((args) => args[0]); + expect(labels).toContain('In-flight subagent refresh failed:'); + expect(labels).not.toContain('Follow-up subagent refresh failed:'); + expect(labels).not.toContain('Subagent refresh failed:'); + const inFlightCall = agentWarnCalls.find( + (args) => args[0] === 'In-flight subagent refresh failed:', + ); + expect((inFlightCall?.[1] as Error).message).toBe( + 'IN-FLIGHT-setTools-rejected', + ); } finally { process.removeListener('unhandledRejection', onUnhandledRejection); } diff --git a/packages/core/src/tools/agent/agent.ts b/packages/core/src/tools/agent/agent.ts index e5183a488be..3eafa723bdc 100644 --- a/packages/core/src/tools/agent/agent.ts +++ b/packages/core/src/tools/agent/agent.ts @@ -927,6 +927,12 @@ export class AgentTool extends BaseDeclarativeTool { * arm exactly one follow-up refresh instead of coalescing into the stale * read. The voided chains are .catch()-guarded per this file's contract: * runRefreshSubagents can reject (its finally awaits llmClient.setTools()). + * The armed chain's .catch() sits BEFORE its .finally(): .finally() + * propagates the original rejection and its own callback is synchronous, so + * a handler placed after it could only ever receive the IN-FLIGHT refresh's + * rejection while labelling it a follow-up failure. The follow-up started + * inside that callback reports its own rejection through the + * requestRefresh() handler below. */ private requestRefresh(): void { if (this.disposed) { @@ -936,13 +942,13 @@ export class AgentTool extends BaseDeclarativeTool { if (!this.listenerRefreshArmed) { this.listenerRefreshArmed = true; void this.refreshInFlight + .catch((error) => + debugLogger.warn('In-flight subagent refresh failed:', error), + ) .finally(() => { this.listenerRefreshArmed = false; this.requestRefresh(); - }) - .catch((error) => - debugLogger.warn('Follow-up subagent refresh failed:', error), - ); + }); } return; } From 4c101ec361dbb2174776f3ca1c9a67ed681b0eb6 Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Tue, 29 Sep 2026 22:06:14 +0800 Subject: [PATCH 09/27] fix(core): treat a target's direct and bridge retry channels as one family The channel marker keeps a bridged pre-check refusal from prefix-pruning the same target's direct validation failure WITHIN a batch, but the batch-start presence prune widened the presence set only for requests carrying an INVALID_TOOL_PARAMS bridge refusal. Each batch therefore preserved just the channel it contained and deleted the other, so a model alternating between bridging a deferred target and calling it directly reset both counters every turn: neither ever reached VALIDATION_RETRY_LOOP_THRESHOLD and RETRY LOOP DETECTED was never injected. That is a regression against the merge base, where both channels produced the same message under the same key and accrued together. Fix both halves, since widening presence alone is not safe: - presence: a request naming X now preserves both `X` and `bridgeRetryToolName(X)`, so alternating channels across batches accrue toward one threshold. - clearing: `clearRetryCountsForTool` now also deletes the bridge-marked keys. A resolved bridge renames the request to the target, so the widened presence set keeps its stale bridge count alive across the successful execution; without the clearing half that resurrects the premature RETRY LOOP DETECTED the marker's prune was relied on to drop. Unrelated tools are unaffected: the widening is keyed on names present in the current batch, and an EXECUTION_DENIED refusal still keeps the wrapper name, so it cannot retain the denied target's counters. Tests: add 'accrues alternating bridged and direct failures of one target across separate batches' (six single-request batches; the directive fires once, on the fifth) and 'clears a target's bridge-marked counter when a bridged call to it succeeds'; extend the existing mixed-channel test to also assert the directive on the bridged half, which its name already claimed. Dropping the presence widening turns the first red; reverting clearRetryCountsForTool to the bare prefix turns the second red. Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-closeout/jmumpkv7rf3 --- .../core/src/core/coreToolScheduler.test.ts | 179 ++++++++++++++++++ packages/core/src/core/coreToolScheduler.ts | 57 ++++-- 2 files changed, 217 insertions(+), 19 deletions(-) diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index 7ea0d3c68a1..c6a680adb38 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -2064,6 +2064,13 @@ describe('CoreToolScheduler', () => { const [bridgeRefusal, directFailure] = third; expect(bridgeRefusal.status).toBe('error'); expect(directFailure.status).toBe('error'); + // Both halves of the claim this test's name makes: each channel reaches + // the threshold on its own key by the third batch. + if (bridgeRefusal.status === 'error') { + expect(bridgeRefusal.response.error?.message).toContain( + 'RETRY LOOP DETECTED', + ); + } if (directFailure.status === 'error') { expect(directFailure.response.error?.message).toContain( 'RETRY LOOP DETECTED', @@ -2071,6 +2078,178 @@ describe('CoreToolScheduler', () => { } }); + it('accrues alternating bridged and direct failures of one target across separate batches', async () => { + // The channel marker alone is not enough: the batch-start presence prune + // runs per batch, so if each batch preserves only the channel it contains, + // a model that alternates between bridging a deferred target and calling + // it directly deletes the OTHER channel's counter every turn. Neither + // counter then exceeds 1, RETRY LOOP DETECTED is never injected, and the + // mixed-channel loop runs indefinitely — the stagnation this PR exists to + // stop. Mutation check: dropping `bridgeRetryToolName(r.name)` from the + // presence set in _schedule turns this red (no batch ever fires). + const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); + const writeFile = new MockTool({ + name: 'write_file', + shouldDefer: true, + params: { + type: 'object', + properties: { + file_path: { type: 'string' }, + content: { type: 'string' }, + }, + required: ['file_path', 'content'], + additionalProperties: false, + }, + }); + const { scheduler, onAllToolCallsComplete } = + createSchedulerForLegacyToolTests({ + toolsByName: new Map([ + [bridge.name, bridge], + [writeFile.name, writeFile], + ]), + deferredHiddenNames: new Set([writeFile.name]), + }); + + // One request per batch, so each batch's presence set holds exactly one + // channel plus whatever the widening adds for it. + const runSingleBatch = async ( + callId: string, + request: { name: string; args: Record }, + ) => { + onAllToolCallsComplete.mockClear(); + await scheduler.schedule( + { + callId, + name: request.name, + args: request.args, + isClientInitiated: false, + prompt_id: 'prompt-bridge-alternating', + }, + new AbortController().signal, + ); + await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); + return onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; + }; + const bridged = { + name: ToolNames.TOOL_CALL, + args: { name: 'write_file', arguments: {} }, + }; + const direct = { name: 'write_file', args: {} }; + + const messages: string[] = []; + for (const batchId of [1, 2, 3, 4, 5, 6]) { + const completed = await runSingleBatch( + `alternating-${batchId}`, + batchId % 2 === 1 ? bridged : direct, + ); + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.errorType).toBe( + ToolErrorType.INVALID_TOOL_PARAMS, + ); + messages.push(completed.response.error?.message ?? ''); + } + } + + // The bridged channel climbs 1, 2, 3 across the odd batches while the + // even batches restart the direct channel, so the directive fires exactly + // once — on the fifth batch — and never prematurely. + expect(messages).toHaveLength(6); + for (const early of messages.slice(0, 4)) { + expect(early).not.toContain('RETRY LOOP DETECTED'); + } + expect(messages[4]).toContain('RETRY LOOP DETECTED'); + expect(messages[5]).not.toContain('RETRY LOOP DETECTED'); + }); + + it('clears a target’s bridge-marked counter when a bridged call to it succeeds', async () => { + // The clearing half of the presence widening. A resolved bridge renames + // the request to the TARGET, so the batch-start presence set now keeps + // both of that target's channels and the prune no longer drops the stale + // bridge-marked count. clearRetryCountsForTool must cover the marked + // channel too, or the surviving count of 2 plus two later refusals would + // fire RETRY LOOP DETECTED prematurely. Mutation check: reverting + // clearRetryCountsForTool to the bare `${toolName}:` prefix turns this red. + const execute = vi.fn().mockResolvedValue({ + llmContent: [{ text: 'published' }], + returnDisplay: 'published', + }); + const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); + // A neutral target: PATH_ARG_KEYS (file_path/path/...) are rewritten on + // request.args during execution, which is not what this case measures. + const publishNote = new MockTool({ + name: 'publish_note', + shouldDefer: true, + execute, + params: { + type: 'object', + properties: { + note_id: { type: 'string' }, + body: { type: 'string' }, + }, + required: ['note_id', 'body'], + additionalProperties: false, + }, + }); + const { scheduler, onAllToolCallsComplete } = + createSchedulerForLegacyToolTests({ + toolsByName: new Map([ + [bridge.name, bridge], + [publishNote.name, publishNote], + ]), + deferredHiddenNames: new Set([publishNote.name]), + }); + + const runBridged = async ( + callId: string, + args: Record, + ) => { + onAllToolCallsComplete.mockClear(); + await scheduler.schedule( + { + callId, + name: ToolNames.TOOL_CALL, + args: { name: 'publish_note', arguments: args }, + isClientInitiated: false, + prompt_id: 'prompt-bridge-clear', + }, + new AbortController().signal, + ); + await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); + return onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; + }; + + // Two bridged refusals take the marked channel to 2. + for (const batchId of [1, 2]) { + const completed = await runBridged(`clear-${batchId}`, {}); + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.error?.message).not.toContain( + 'RETRY LOOP DETECTED', + ); + } + } + + // A successful bridged execution of the SAME target clears both channels. + const succeeded = await runBridged('clear-success', { + note_id: 'n-1', + body: 'ok', + }); + expect(succeeded.status).toBe('success'); + expect(execute).toHaveBeenCalledTimes(1); + + // Two more refusals restart at 1 instead of inheriting the stale count. + for (const batchId of [3, 4]) { + const completed = await runBridged(`clear-after-${batchId}`, {}); + expect(completed.status).toBe('error'); + if (completed.status === 'error') { + expect(completed.response.error?.message).not.toContain( + 'RETRY LOOP DETECTED', + ); + } + } + }); + it('applies the retry-loop directive to repeated invalid tool_call envelopes', async () => { const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); const { scheduler, onAllToolCallsComplete } = diff --git a/packages/core/src/core/coreToolScheduler.ts b/packages/core/src/core/coreToolScheduler.ts index 5dbabd32793..05859b33b50 100644 --- a/packages/core/src/core/coreToolScheduler.ts +++ b/packages/core/src/core/coreToolScheduler.ts @@ -2882,12 +2882,15 @@ export class CoreToolScheduler { /** * Removes all validation retry counters for the given tool. Keys are * ":", so a plain `Map.delete(toolName)` would not - * match anything. + * match anything. The bridge-marked channel is cleared too: the two channels + * are one family for presence (see the prune in _schedule), so clearing must + * cover both or a successful execution of the target would leave its stale + * bridge-channel count behind to fire RETRY LOOP DETECTED prematurely. */ private clearRetryCountsForTool(toolName: string): void { - const prefix = `${toolName}:`; + const prefixes = [`${toolName}:`, `${bridgeRetryToolName(toolName)}:`]; for (const key of this.validationRetryCounts.keys()) { - if (key.startsWith(prefix)) { + if (prefixes.some((prefix) => key.startsWith(prefix))) { this.validationRetryCounts.delete(key); } } @@ -2972,24 +2975,40 @@ export class CoreToolScheduler { // present in the current batch. Keeping every tracked tool's counters // whenever any current request matched caused stale counts for // unrelated tools to survive and fire RETRY LOOP DETECTED prematurely - // the next time those tools were used. A refused bridge request keeps - // the wrapper name (`tool_call`), so the channel-marked name of the - // validated target must join the presence set alongside it — but only - // for INVALID_TOOL_PARAMS refusals, the one error type that accrues - // below: an EXECUTION_DENIED (policy) refusal records nothing, so it - // must not keep the denied target's stale counters alive either. + // the next time those tools were used. + // + // A target's direct and bridge-marked channels are ONE family for + // presence: a request naming X preserves both `X` and + // `bridgeRetryToolName(X)`. Widening presence only for the bridged + // channel let alternating channels ACROSS batches prune each other — a + // bridged batch kept just the marked key and a direct batch just the + // bare one, so neither counter ever reached + // VALIDATION_RETRY_LOOP_THRESHOLD and a mixed-channel loop rode on + // without the stop directive. clearRetryCountsForTool clears both + // channels, so this widening cannot resurrect the stale bridge count + // that a resolved-and-executed target leaves behind. + // + // A refused bridge request keeps the wrapper name (`tool_call`), so the + // channel-marked name of the validated target must join the presence set + // alongside it — but only for INVALID_TOOL_PARAMS refusals, the one error + // type that accrues below: an EXECUTION_DENIED (policy) refusal records + // nothing, so it must not keep the denied target's stale counters alive + // either. if (this.validationRetryCounts.size > 0) { const currentToolNames = new Set( - requestsToProcess.flatMap((r) => - r.bridgeResolutionError?.type === - ToolErrorType.INVALID_TOOL_PARAMS && - r.bridgeResolutionError.targetName !== undefined - ? [ - r.name, - bridgeRetryToolName(r.bridgeResolutionError.targetName), - ] - : [r.name], - ), + requestsToProcess.flatMap((r) => { + const names = [r.name, bridgeRetryToolName(r.name)]; + if ( + r.bridgeResolutionError?.type === + ToolErrorType.INVALID_TOOL_PARAMS && + r.bridgeResolutionError.targetName !== undefined + ) { + names.push( + bridgeRetryToolName(r.bridgeResolutionError.targetName), + ); + } + return names; + }), ); for (const key of [...this.validationRetryCounts.keys()]) { const sep = key.indexOf(':'); From 2059aea6dd525c5b323ae4622aebe1ee5c061e82 Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Tue, 29 Sep 2026 22:16:24 +0800 Subject: [PATCH 10/27] fix(cli): let ACP's own L1 gate own a policy-denied bridge target MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ACP wired the permission manager into isTargetExecutionAllowed, but that option is the outer owner's execution-allowlist hook — only agent-core supplies one, and this frontend reads neither an execution allowlist nor a disallowedTools blocklist. So with `permissions.deny: ["web_fetch"]` driven through ACP, resolution short-circuited with the bridge's scheduler-flavoured wording ("execution allowlist or disallowedTools blocklist"), pointing the operator at two knobs that cannot be the cause and never naming the deny rule, while the identical configuration on the terminal frontend prints `Matching deny rule: "permissions.deny: web_fetch"`. Mirror the scheduler's split instead: the permission manager goes into suppressArgumentPreCheck, so a pm-denied target skips the argument pre-check (no INVALID_TOOL_PARAMS strike for a call that could never run) and the L1 enablement gate below stays ACP's only policy authority — carrying ACP's own denial wording and its isTrustedLiveTool exemption, exactly as a direct call to the same target already does. The load-bearing invariants are unchanged: the call is still denied before execution, still EXECUTION_DENIED, the target is still never built, and no invalid-parameter strike is recorded. Test: retarget `keeps EXECUTION_DENIED for a policy-denied bridge target instead of a parameter pre-check refusal` from the bridge wording to ACP's own `Tool "web_fetch" is disabled.`, and assert the bridge wording is gone. Reverting the wiring turns it red. Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-closeout/jmumpkv7rf3 --- .../acp-integration/session/Session.test.ts | 31 ++++++++++++------- .../src/acp-integration/session/Session.ts | 19 +++++++++--- 2 files changed, 33 insertions(+), 17 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.test.ts b/packages/cli/src/acp-integration/session/Session.test.ts index b30c38eede4..54ac29c3f81 100644 --- a/packages/cli/src/acp-integration/session/Session.test.ts +++ b/packages/cli/src/acp-integration/session/Session.test.ts @@ -18465,14 +18465,20 @@ describe('Session', () => { it('keeps EXECUTION_DENIED for a policy-denied bridge target instead of a parameter pre-check refusal', async () => { // The ACP half of the resolution-time policy wiring: with the - // wrapper allowed but the target denied, resolution must refuse with - // the policy denial BEFORE the argument pre-check — otherwise a - // denied target with malformed arguments gets INVALID_TOOL_PARAMS - // plus an invalid-parameter strike toward the loop stop for a call - // that could never run. Mutation check: removing the - // `...(pm ? { isTargetExecutionAllowed } : {})` spread in - // Session.runTool turns this red (errorType flips to - // INVALID_TOOL_PARAMS and invalidToolParamErrors gains an entry). + // wrapper allowed but the target denied, the denial must stay with + // ACP's own L1 enablement gate and must not become a parameter + // pre-check refusal — otherwise a denied target with malformed + // arguments gets INVALID_TOOL_PARAMS plus an invalid-parameter strike + // toward the loop stop for a call that could never run. The pm is + // wired into suppressArgumentPreCheck (not isTargetExecutionAllowed, + // which is the outer owner's execution-allowlist hook that this + // frontend does not have), so the pre-check is skipped for a denied + // target and L1 supplies both the wording and the isTrustedLiveTool + // exemption. Mutation check: removing the + // `...(pm ? { suppressArgumentPreCheck } : {})` spread in + // Session.runTool turns this red (the denial becomes an + // INVALID_TOOL_PARAMS bridge refusal naming `required property 'url'` + // and invalidToolParamErrors gains an entry). mockConfig.getApprovalMode = vi.fn().mockReturnValue(ApprovalMode.YOLO); mockConfig.getPermissionManager = vi.fn().mockReturnValue({ isToolEnabled: vi.fn(async (name: string) => name !== 'web_fetch'), @@ -18560,10 +18566,11 @@ describe('Session', () => { const errorText = String( result.parts[0]?.functionResponse?.response?.['error'], ); - expect( - errorText.startsWith(core.DEFERRED_TOOL_CALL_REFUSAL_PREFIX), - ).toBe(true); - expect(errorText).toContain( + // ACP's own L1 wording, not the bridge's scheduler-flavoured message + // that points an operator at an execution allowlist / disallowedTools + // blocklist this frontend never reads. + expect(errorText).toContain('Tool "web_fetch" is disabled.'); + expect(errorText).not.toContain( "not permitted by this agent's tool policy", ); expect(errorText).not.toContain("required property 'url'"); diff --git a/packages/cli/src/acp-integration/session/Session.ts b/packages/cli/src/acp-integration/session/Session.ts index 001e79bdcf0..2394e92a928 100644 --- a/packages/cli/src/acp-integration/session/Session.ts +++ b/packages/cli/src/acp-integration/session/Session.ts @@ -13546,13 +13546,22 @@ export class Session implements SessionContext { // all three frontends (wenshao triage follow-up). Omitting it would // fail closed, not open, but the corner case should agree everywhere. maxSubagentDepth: this.config.getMaxSubagentDepth(), - // The L1 enablement gate below runs after resolution; consult the - // same policy inside resolution so a denied target keeps its - // EXECUTION_DENIED instead of a parameter pre-check refusal. + // Mirror the scheduler's split of the two bridge policy options: + // isTargetExecutionAllowed is the OUTER OWNER's execution allowlist + // (only agent-core supplies one — this frontend has no such knob), + // while the permission manager owns enablement / deny rules. So the pm + // goes into suppressArgumentPreCheck: a pm-denied target skips the + // argument pre-check (no INVALID_TOOL_PARAMS strike for a call that + // could never run) and the L1 enablement gate below stays ACP's only + // policy authority, keeping its own denial wording and its + // isTrustedLiveTool exemption. Wiring the pm into + // isTargetExecutionAllowed instead made resolution short-circuit with + // the bridge's scheduler-flavoured message, which names an execution + // allowlist and a disallowedTools blocklist that ACP never reads. ...(pm ? { - isTargetExecutionAllowed: (targetName: string) => - pm.isToolEnabled(targetName), + suppressArgumentPreCheck: async (targetName: string) => + !(await pm.isToolEnabled(targetName)), } : {}), }); From 29d3080cf38dc592f5ff196bd5710b7496909db1 Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Wed, 30 Sep 2026 00:14:15 +0800 Subject: [PATCH 11/27] fix(core): preserve target validation and contain bridge policy errors --- .../acp-integration/session/Session.test.ts | 262 ++++++++++-------- .../src/acp-integration/session/Session.ts | 9 +- packages/core/src/tools/agent/agent.test.ts | 233 ---------------- packages/core/src/tools/agent/agent.ts | 67 +---- packages/core/src/tools/tool-call.test.ts | 75 ++++- packages/core/src/tools/tool-call.ts | 13 +- 6 files changed, 243 insertions(+), 416 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.test.ts b/packages/cli/src/acp-integration/session/Session.test.ts index 54ac29c3f81..f1810bdff5a 100644 --- a/packages/cli/src/acp-integration/session/Session.test.ts +++ b/packages/cli/src/acp-integration/session/Session.test.ts @@ -18463,120 +18463,162 @@ describe('Session', () => { ); }); - it('keeps EXECUTION_DENIED for a policy-denied bridge target instead of a parameter pre-check refusal', async () => { - // The ACP half of the resolution-time policy wiring: with the - // wrapper allowed but the target denied, the denial must stay with - // ACP's own L1 enablement gate and must not become a parameter - // pre-check refusal — otherwise a denied target with malformed - // arguments gets INVALID_TOOL_PARAMS plus an invalid-parameter strike - // toward the loop stop for a call that could never run. The pm is - // wired into suppressArgumentPreCheck (not isTargetExecutionAllowed, - // which is the outer owner's execution-allowlist hook that this - // frontend does not have), so the pre-check is skipped for a denied - // target and L1 supplies both the wording and the isTrustedLiveTool - // exemption. Mutation check: removing the - // `...(pm ? { suppressArgumentPreCheck } : {})` spread in - // Session.runTool turns this red (the denial becomes an - // INVALID_TOOL_PARAMS bridge refusal naming `required property 'url'` - // and invalidToolParamErrors gains an entry). - mockConfig.getApprovalMode = vi.fn().mockReturnValue(ApprovalMode.YOLO); - mockConfig.getPermissionManager = vi.fn().mockReturnValue({ - isToolEnabled: vi.fn(async (name: string) => name !== 'web_fetch'), - }); - const bridge = { - name: core.ToolNames.TOOL_CALL, - kind: core.Kind.Other, - description: 'Deferred tool bridge', - build: vi.fn((params: Record) => ({ params })), - }; - const toolSearch = { - name: core.ToolNames.TOOL_SEARCH, - kind: core.Kind.Other, - description: 'Deferred tool discovery', - build: vi.fn((params: Record) => ({ params })), - }; - const target = { - name: 'web_fetch', - kind: core.Kind.Other, - description: 'Fetches a URL', - schema: { - parametersJsonSchema: { - type: 'object', - properties: { - url: { type: 'string' }, - prompt: { type: 'string' }, + it.each(['denied', 'unavailable'] as const)( + 'keeps a %s target policy failure within its ACP call', + async (policyState) => { + // The ACP half of the resolution-time policy wiring: with the + // wrapper allowed but the target denied, the denial must stay with + // ACP's own L1 enablement gate and must not become a parameter + // pre-check refusal — otherwise a denied target with malformed + // arguments gets INVALID_TOOL_PARAMS plus an invalid-parameter strike + // toward the loop stop for a call that could never run. The pm is + // wired into suppressArgumentPreCheck (not isTargetExecutionAllowed, + // which is the outer owner's execution-allowlist hook that this + // frontend does not have), so the pre-check is skipped for a denied + // target and L1 supplies both the wording and the isTrustedLiveTool + // exemption. Mutation check: removing the + // `...(pm ? { suppressArgumentPreCheck } : {})` spread in + // Session.runTool turns this red (the denial becomes an + // INVALID_TOOL_PARAMS bridge refusal naming `required property 'url'` + // and invalidToolParamErrors gains an entry). + mockConfig.getApprovalMode = vi + .fn() + .mockReturnValue(ApprovalMode.YOLO); + mockConfig.getPermissionManager = vi.fn().mockReturnValue({ + isToolEnabled: vi.fn(async (name: string) => { + if (name !== 'web_fetch') return true; + if (policyState === 'unavailable') + throw new Error('policy store unavailable'); + return false; + }), + }); + const bridge = { + name: core.ToolNames.TOOL_CALL, + kind: core.Kind.Other, + description: 'Deferred tool bridge', + build: vi.fn((params: Record) => ({ params })), + }; + const toolSearch = { + name: core.ToolNames.TOOL_SEARCH, + kind: core.Kind.Other, + description: 'Deferred tool discovery', + build: vi.fn((params: Record) => ({ params })), + }; + const target = { + name: 'web_fetch', + kind: core.Kind.Other, + description: 'Fetches a URL', + schema: { + parametersJsonSchema: { + type: 'object', + properties: { + url: { type: 'string' }, + prompt: { type: 'string' }, + }, + required: ['url', 'prompt'], + additionalProperties: false, }, - required: ['url', 'prompt'], - additionalProperties: false, }, - }, - build: vi.fn(), - }; - mockToolRegistry.getTool.mockImplementation((name: string) => - name === bridge.name - ? bridge - : name === target.name - ? target - : name === toolSearch.name - ? toolSearch - : undefined, - ); - mockToolRegistry.ensureTool.mockImplementation(async (name: string) => - name === bridge.name - ? bridge - : name === target.name - ? target - : name === toolSearch.name - ? toolSearch - : undefined, - ); - mockToolRegistry.isDeferredAndHidden.mockImplementation( - (name: string) => name === target.name, - ); - const toolLoopState = { - totalToolCalls: 0, - invalidToolParamErrors: new Map(), - toolCallKeyCounts: new Map(), - maxToolCallKeyRepeat: 0, - loopDetected: false, - }; + build: vi.fn(), + }; + const sibling = { + name: 'sibling_tool', + kind: core.Kind.Other, + displayName: 'Sibling', + description: 'An independent call', + build: vi.fn((params: Record) => ({ + params, + getDefaultPermission: vi.fn().mockResolvedValue('allow'), + getDescription: () => 'independent call', + toolLocations: () => [], + execute: vi + .fn() + .mockResolvedValue({ + llmContent: 'sibling completed', + returnDisplay: 'sibling completed', + }), + })), + }; + mockToolRegistry.getTool.mockImplementation((name: string) => + name === bridge.name + ? bridge + : name === target.name + ? target + : name === toolSearch.name + ? toolSearch + : name === sibling.name + ? sibling + : undefined, + ); + mockToolRegistry.ensureTool.mockImplementation( + async (name: string) => + name === bridge.name + ? bridge + : name === target.name + ? target + : name === toolSearch.name + ? toolSearch + : undefined, + ); + mockToolRegistry.isDeferredAndHidden.mockImplementation( + (name: string) => name === target.name, + ); + const toolLoopState = { + totalToolCalls: 0, + invalidToolParamErrors: new Map(), + toolCallKeyCounts: new Map(), + maxToolCallKeyRepeat: 0, + loopDetected: false, + }; - const result = await ( - session as unknown as { - runToolCalls: ( - abortSignal: AbortSignal, - promptId: string, - calls: FunctionCall[], - loopState: typeof toolLoopState, - ) => Promise<{ parts: Part[] }>; - } - ).runToolCalls( - new AbortController().signal, - 'prompt-tool-call-bridge-denied', - [ - { - id: 'bridge-denied-call', - name: core.ToolNames.TOOL_CALL, - args: { name: target.name, arguments: {} }, - }, - ], - toolLoopState, - ); + const result = await ( + session as unknown as { + runToolCalls: ( + abortSignal: AbortSignal, + promptId: string, + calls: FunctionCall[], + loopState: typeof toolLoopState, + ) => Promise<{ parts: Part[] }>; + } + ).runToolCalls( + new AbortController().signal, + 'prompt-tool-call-bridge-denied', + [ + { + id: 'bridge-denied-call', + name: core.ToolNames.TOOL_CALL, + args: { name: target.name, arguments: {} }, + }, + { id: 'independent-call', name: sibling.name, args: {} }, + ], + toolLoopState, + ); - const errorText = String( - result.parts[0]?.functionResponse?.response?.['error'], - ); - // ACP's own L1 wording, not the bridge's scheduler-flavoured message - // that points an operator at an execution allowlist / disallowedTools - // blocklist this frontend never reads. - expect(errorText).toContain('Tool "web_fetch" is disabled.'); - expect(errorText).not.toContain( - "not permitted by this agent's tool policy", - ); - expect(errorText).not.toContain("required property 'url'"); - expect(toolLoopState.invalidToolParamErrors.size).toBe(0); - expect(target.build).not.toHaveBeenCalled(); - }); + const errorText = String( + result.parts[0]?.functionResponse?.response?.['error'], + ); + // ACP's own L1 wording, not the bridge's scheduler-flavoured message + // that points an operator at an execution allowlist / disallowedTools + // blocklist this frontend never reads. + expect(errorText).toContain( + policyState === 'denied' + ? 'Tool "web_fetch" is disabled.' + : 'policy store unavailable', + ); + expect(errorText).not.toContain( + "not permitted by this agent's tool policy", + ); + expect(errorText).not.toContain("required property 'url'"); + expect(toolLoopState.invalidToolParamErrors.size).toBe(0); + expect(target.build).not.toHaveBeenCalled(); + expect(result.parts).toHaveLength(2); + expect(result.parts[1]?.functionResponse).toMatchObject({ + id: 'independent-call', + name: sibling.name, + response: { output: 'sibling completed' }, + }); + }, + ); it('marks a disabled ACP tool_call as a bridge refusal', async () => { mockConfig.getPermissionManager = vi.fn().mockReturnValue({ diff --git a/packages/cli/src/acp-integration/session/Session.ts b/packages/cli/src/acp-integration/session/Session.ts index 2394e92a928..da86dd35e0c 100644 --- a/packages/cli/src/acp-integration/session/Session.ts +++ b/packages/cli/src/acp-integration/session/Session.ts @@ -13564,7 +13564,14 @@ export class Session implements SessionContext { !(await pm.isToolEnabled(targetName)), } : {}), - }); + }).catch( + ( + error: unknown, + ): Awaited> => ({ + error: error instanceof Error ? error : new Error(String(error)), + errorType: ToolErrorType.UNHANDLED_EXCEPTION, + }), + ); const bridgeCancellation = cancelBeforeExecutionIfAborted(toolName); if (bridgeCancellation) return bridgeCancellation; if ('error' in resolution) { diff --git a/packages/core/src/tools/agent/agent.test.ts b/packages/core/src/tools/agent/agent.test.ts index d246052dd1b..803bd8a748a 100644 --- a/packages/core/src/tools/agent/agent.test.ts +++ b/packages/core/src/tools/agent/agent.test.ts @@ -91,29 +91,6 @@ function escapeRegExp(value: string): string { vi.mock('../../subagents/subagent-manager.js'); vi.mock('../../agents/runtime/agent-headless.js'); -// Records the AGENT-tagged debugLogger.warn() calls so tests can assert which -// refresh-failure label a rejection was reported under. Wraps the real logger -// instead of replacing it, so nothing else in this file changes behaviour — -// same shape as goals/goal-persistence.test.ts. -const agentWarnCalls = vi.hoisted(() => [] as unknown[][]); -vi.mock('../../utils/debugLogger.js', async (importOriginal) => { - const original = - await importOriginal(); - return { - ...original, - createDebugLogger: (tag?: string) => { - const logger = original.createDebugLogger(tag); - return { - ...logger, - warn: (...args: unknown[]) => { - if (tag === 'AGENT') agentWarnCalls.push(args); - logger.warn(...args); - }, - }; - }, - }; -}); - // Spies for the subagent-span layer so tests can assert what status taxonomy // was published. The real runInSubagentSpanContext sets up OTel context-with, // which is irrelevant here — we just need the body to run. Review wenshao @@ -1479,42 +1456,6 @@ describe('AgentTool', () => { expect(result).toBeNull(); }); - it('coalesces a synchronous refresh kick onto the in-flight refresh and arms one follow-up', async () => { - // An unknown subagent_type kicks a refresh so the cache catches up — - // the kick is the ONLY discovery path for agent files created out of - // band (the change listener does not fire for those), and the - // in-flight scan may have read the directory before the file existed, - // so a kick landing mid-refresh must queue one follow-up scan instead - // of being dropped. The second synchronous kick still coalesces onto - // the in-flight scan (one scan, one arm — not three scans). - await agentTool.refreshSubagents(); - const listSpy = vi.mocked(mockSubagentManager.listSubagents); - listSpy.mockClear(); - - agentTool.validateToolParams({ - ...validParams, - subagent_type: 'missing', - }); - agentTool.validateToolParams({ - ...validParams, - subagent_type: 'missing', - }); - expect(listSpy).toHaveBeenCalledTimes(1); - - // The armed follow-up re-scans once the in-flight refresh settles — - // no third kick needed. Mutation check: kicking plain - // refreshSubagents() (dropping the arm) leaves this at 1. - await vi.runAllTimersAsync(); - expect(listSpy).toHaveBeenCalledTimes(2); - - // After everything settled, a later kick re-scans again. - agentTool.validateToolParams({ - ...validParams, - subagent_type: 'missing', - }); - expect(listSpy).toHaveBeenCalledTimes(3); - }); - it('should reject empty description', async () => { const result = agentTool.validateToolParams({ ...validParams, @@ -2888,180 +2829,6 @@ describe('AgentTool', () => { expect(agentTool.description).toContain('A brand new agent'); }); - it('arms exactly one follow-up scan when a change lands mid-refresh', async () => { - // A change landing mid-refresh is not reflected in the in-flight scan - // (it may have read the directory before the file existed), so the - // listener arms one follow-up instead of coalescing into the stale - // read — and two change events during one refresh must yield ONE - // follow-up, not two. Mutation checks: coalescing listener kicks like - // validation kicks (bare return when a refresh is in flight) leaves - // the count at 1; dropping the arm-once guard yields 3. - const listSpy = vi.mocked(mockSubagentManager.listSubagents); - listSpy.mockClear(); - let releaseScan!: () => void; - listSpy.mockImplementationOnce( - () => - new Promise((resolve) => { - releaseScan = () => resolve(mockSubagents); - }), - ); - - void agentTool.refreshSubagents(); - expect(listSpy).toHaveBeenCalledTimes(1); - - const listener = changeListeners[0]; - listener?.(); - listener?.(); - // Both change events coalesced into one armed follow-up; the - // in-flight scan is still the only scan so far. - expect(listSpy).toHaveBeenCalledTimes(1); - - releaseScan(); - await vi.runAllTimersAsync(); - expect(listSpy).toHaveBeenCalledTimes(2); - }); - - it('does not run an armed follow-up refresh after dispose', async () => { - // dispose() removes the change listener, but an already-armed - // follow-up is a separate promise chain: without a disposal guard it - // would still run a full rescan + llmClient.setTools() on a torn-down - // tool. Mutation check: dropping the disposed check from the armed - // callback turns this red (a second scan runs). - const listSpy = vi.mocked(mockSubagentManager.listSubagents); - listSpy.mockClear(); - let releaseScan!: () => void; - listSpy.mockImplementationOnce( - () => - new Promise((resolve) => { - releaseScan = () => resolve(mockSubagents); - }), - ); - - void agentTool.refreshSubagents(); - expect(listSpy).toHaveBeenCalledTimes(1); - - changeListeners[0]?.(); - agentTool.dispose(); - - releaseScan(); - await vi.runAllTimersAsync(); - expect(listSpy).toHaveBeenCalledTimes(1); - }); - - it('absorbs a setTools rejection from an armed follow-up refresh', async () => { - // runRefreshSubagents can reject (its finally awaits - // llmClient.setTools()), and .finally() propagates that rejection into - // the armed chain — without a .catch at the void boundary it floats as - // an unhandledRejection, which this file's own contract forbids. The - // follow-up scan must still run. Mutation checks: dropping the armed - // chain's .catch(...) turns the rejection assertion red, and moving it - // back AFTER the .finally() turns the label assertions red (the - // in-flight rejection would be reported as a follow-up failure). - agentWarnCalls.length = 0; - const setTools = vi.fn().mockRejectedValue(new Error('setTools failed')); - vi.mocked(config.getLlmClient).mockReturnValue({ - setTools, - } as unknown as ReturnType); - const unhandledRejections: unknown[] = []; - const onUnhandledRejection = (reason: unknown) => { - unhandledRejections.push(reason); - }; - process.on('unhandledRejection', onUnhandledRejection); - try { - const listSpy = vi.mocked(mockSubagentManager.listSubagents); - listSpy.mockClear(); - let releaseScan!: () => void; - listSpy.mockImplementationOnce( - () => - new Promise((resolve) => { - releaseScan = () => resolve(mockSubagents); - }), - ); - - // The in-flight refresh rejects in its finally (setTools); the test - // swallows that one directly so only the armed chain is measured. - const inFlight = agentTool.refreshSubagents(); - void inFlight.catch(() => {}); - changeListeners[0]?.(); - - releaseScan(); - await vi.runAllTimersAsync(); - - // The armed follow-up still ran (and hit the rejecting setTools - // again), and no rejection escaped either void boundary. - expect(listSpy).toHaveBeenCalledTimes(2); - expect(unhandledRejections).toEqual([]); - // Each rejection is reported under its own label: the in-flight one by - // the armed chain's .catch, the follow-up's own by requestRefresh's. - const labels = agentWarnCalls.map((args) => args[0]); - expect(labels).toContain('In-flight subagent refresh failed:'); - expect(labels).toContain('Subagent refresh failed:'); - expect(labels).not.toContain('Follow-up subagent refresh failed:'); - } finally { - process.removeListener('unhandledRejection', onUnhandledRejection); - } - }); - - it('reports an in-flight refresh rejection under the in-flight label only', async () => { - // The label half of the armed chain: .finally() propagates the ORIGINAL - // rejection and its callback is synchronous, so a .catch placed after - // it can only ever see the in-flight refresh's error — and would name - // it "Follow-up ... failed" even when the follow-up scan succeeded. - // Here setTools rejects once (the in-flight refresh) and then resolves - // (the follow-up). Mutation check: reordering the armed chain back to - // .finally(...).catch('Follow-up subagent refresh failed:') turns the - // in-flight label assertion red. - agentWarnCalls.length = 0; - const setTools = vi - .fn() - .mockRejectedValueOnce(new Error('IN-FLIGHT-setTools-rejected')) - .mockResolvedValue(undefined); - vi.mocked(config.getLlmClient).mockReturnValue({ - setTools, - } as unknown as ReturnType); - const unhandledRejections: unknown[] = []; - const onUnhandledRejection = (reason: unknown) => { - unhandledRejections.push(reason); - }; - process.on('unhandledRejection', onUnhandledRejection); - try { - const listSpy = vi.mocked(mockSubagentManager.listSubagents); - listSpy.mockClear(); - let releaseScan!: () => void; - listSpy.mockImplementationOnce( - () => - new Promise((resolve) => { - releaseScan = () => resolve(mockSubagents); - }), - ); - - const inFlight = agentTool.refreshSubagents(); - void inFlight.catch(() => {}); - changeListeners[0]?.(); - - releaseScan(); - await vi.runAllTimersAsync(); - - // The follow-up scan ran and its setTools succeeded. - expect(listSpy).toHaveBeenCalledTimes(2); - expect(setTools).toHaveBeenCalledTimes(2); - expect(unhandledRejections).toEqual([]); - - const labels = agentWarnCalls.map((args) => args[0]); - expect(labels).toContain('In-flight subagent refresh failed:'); - expect(labels).not.toContain('Follow-up subagent refresh failed:'); - expect(labels).not.toContain('Subagent refresh failed:'); - const inFlightCall = agentWarnCalls.find( - (args) => args[0] === 'In-flight subagent refresh failed:', - ); - expect((inFlightCall?.[1] as Error).message).toBe( - 'IN-FLIGHT-setTools-rejected', - ); - } finally { - process.removeListener('unhandledRejection', onUnhandledRejection); - } - }); - it('should refresh available subagents and update description', async () => { const newSubagents: SubagentConfig[] = [ { diff --git a/packages/core/src/tools/agent/agent.ts b/packages/core/src/tools/agent/agent.ts index 3eafa723bdc..a2f659db3f7 100644 --- a/packages/core/src/tools/agent/agent.ts +++ b/packages/core/src/tools/agent/agent.ts @@ -886,7 +886,7 @@ export class AgentTool extends BaseDeclarativeTool { this.delegationSurface = resolveAgentDelegationSurface(config); this.subagentManager = config.getSubagentManager(); this.removeChangeListener = this.subagentManager.addChangeListener(() => { - this.requestRefresh(); + void this.refreshSubagents(); }); // Initialize the tool asynchronously @@ -894,70 +894,14 @@ export class AgentTool extends BaseDeclarativeTool { } dispose(): void { - this.disposed = true; this.removeChangeListener(); } - private refreshInFlight: Promise | undefined; - private listenerRefreshArmed = false; - private disposed = false; - /** * Asynchronously initializes the tool by loading available subagents * and updating the description and schema. - * - * Concurrent callers coalesce onto the in-flight refresh: a second scan + - * setTools mid-turn buys nothing on its own. Signals that mean "subagent - * state changed" (the change listener, the unknown-subagent_type - * validation kick) go through requestRefresh() instead, because the - * in-flight scan may have read the directory before the change landed. - */ - refreshSubagents(): Promise { - this.refreshInFlight ??= this.runRefreshSubagents().finally(() => { - this.refreshInFlight = undefined; - }); - return this.refreshInFlight; - } - - /** - * Fire-and-forget refresh for out-of-band signals (the change listener and - * the validation kick — the kick is the only discovery path for agent - * files created outside the manager, which the listener never sees). A - * signal landing mid-refresh is not reflected in the in-flight scan, so - * arm exactly one follow-up refresh instead of coalescing into the stale - * read. The voided chains are .catch()-guarded per this file's contract: - * runRefreshSubagents can reject (its finally awaits llmClient.setTools()). - * The armed chain's .catch() sits BEFORE its .finally(): .finally() - * propagates the original rejection and its own callback is synchronous, so - * a handler placed after it could only ever receive the IN-FLIGHT refresh's - * rejection while labelling it a follow-up failure. The follow-up started - * inside that callback reports its own rejection through the - * requestRefresh() handler below. */ - private requestRefresh(): void { - if (this.disposed) { - return; - } - if (this.refreshInFlight) { - if (!this.listenerRefreshArmed) { - this.listenerRefreshArmed = true; - void this.refreshInFlight - .catch((error) => - debugLogger.warn('In-flight subagent refresh failed:', error), - ) - .finally(() => { - this.listenerRefreshArmed = false; - this.requestRefresh(); - }); - } - return; - } - void this.refreshSubagents().catch((error) => - debugLogger.warn('Subagent refresh failed:', error), - ); - } - - private async runRefreshSubagents(): Promise { + async refreshSubagents(): Promise { try { this.availableSubagents = await this.subagentManager.listSubagents(); this.updateDescriptionAndSchema(); @@ -1183,11 +1127,8 @@ The background-agent rules above apply to background forks unchanged.${delegatio // resolves the type via loadSubagent(), which reads from disk and // fails with a clear "not found" error if the agent truly doesn't // exist. Kick a refresh (validation must stay synchronous) so the - // cache and schema catch up for subsequent calls. The kick goes - // through requestRefresh(): a kick landing mid-scan arms one - // follow-up, because the in-flight scan may have read the - // directory before the file existed. - this.requestRefresh(); + // cache and schema catch up for subsequent calls. + void this.refreshSubagents(); } } } diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 4bc500f3e8d..0b4e993e06f 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -816,6 +816,64 @@ describe('ToolCallTool', () => { } }); + it('leaves surplus-key enforcement to the target without changing its schema', async () => { + class LenientTool extends MockTool { + override validateToolParams(): string | null { + return null; + } + } + for (const Tool of [MockTool, LenientTool]) { + const target = new Tool({ + name: 'agent_like', + shouldDefer: true, + params: { + type: 'object', + properties: { prompt: { type: 'string' } }, + required: ['prompt'], + additionalProperties: false, + }, + }); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { + name: target.name, + arguments: { prompt: 'investigate', name: 'helper' }, + }, + ); + expect(result).not.toHaveProperty('error'); + expect(target.schema.parametersJsonSchema).toHaveProperty( + 'additionalProperties', + false, + ); + if ('tool' in result) { + const build = () => result.tool.build(result.arguments); + if (Tool === MockTool) { + expect(build).toThrow('must NOT have additional properties'); + } else { + expect(build).not.toThrow(); + } + } + } + }); + + it('still attributes wrong field types when surplus keys are present', async () => { + const target = makeWebFetchLike(); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { + name: target.name, + arguments: { url: {}, prompt: 'summarize', extra: true }, + }, + ); + expect(result).toMatchObject({ + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: 'web_fetch', + error: expect.objectContaining({ + message: expect.stringContaining('must be string'), + }), + }); + }); + it('returns the model-sent arguments even when validation coerces a clone', async () => { // SchemaValidator.validate coerces values in place (numeric strings → // numbers, etc.). The pre-check must run on a clone: the resolved @@ -843,11 +901,9 @@ describe('ToolCallTool', () => { // (it adds and removes `model`/`name`), and Ajv caches a compiled // schema by object identity for the life of the process. Handing the // validator that same object pins every later bridged call to the - // shape the first one happened to compile, so a property the target - // advertises after that first call is refused forever. - // Mutation check: passing target.schema.parametersJsonSchema by - // reference instead of a clone turns this red with "must NOT have - // additional properties". + // shape the first one happened to compile, ignoring changed constraints. + // A stale validator would ignore the newly advertised model enum and + // accept the unknown grade below. const target = new MockTool({ name: 'agent_like', shouldDefer: true, @@ -872,6 +928,15 @@ describe('ToolCallTool', () => { }; schema.properties['model'] = { type: 'string', enum: ['fast', 'pro'] }; + const invalid = await resolveDeferredToolCall(registry, { + name: 'agent_like', + arguments: { prompt: 'do the thing', model: 'unknown-grade' }, + }); + expect(invalid).toMatchObject({ + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: 'agent_like', + }); + const result = await resolveDeferredToolCall(registry, { name: 'agent_like', arguments: { prompt: 'do the thing', model: 'fast' }, diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index fd014a8050c..1ee06b092f6 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -310,10 +310,15 @@ export async function resolveDeferredToolCall( if (!options?.wasOutputTruncated && !preCheckSuppressed) { try { const argsClone = structuredClone(invocation.params.arguments); - paramsError = SchemaValidator.validate( - structuredClone(target.schema.parametersJsonSchema), - argsClone, - ); + const schemaClone = structuredClone( + target.schema.parametersJsonSchema, + ) as Record; + // Some targets deliberately tolerate surplus keys (for example, Agent's + // name outside team mode). Leave that decision to their own build(). + if (schemaClone['additionalProperties'] === false) { + schemaClone['additionalProperties'] = true; + } + paramsError = SchemaValidator.validate(schemaClone, argsClone); } catch { // A target whose validation throws under this pre-check must not become // a new bridge failure mode: the scheduler's build() reports the same From 6c9d70160febd2ed8a6854c2d000e943f6bd3d13 Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Wed, 30 Sep 2026 02:53:05 +0800 Subject: [PATCH 12/27] test(cli): complete the pm stub in the ACP target-policy bridge tests The new `keeps a %s target policy failure within its ACP call` cases stub `getPermissionManager` with only `isToolEnabled`. The independent sibling call in the same batch still reaches `evaluatePermissionRules`, which calls `pm.hasRelevantRules()` whenever the tool's own default permission is not `deny` -- and the sibling declares `getDefaultPermission: 'allow'`. The missing method threw, so the sibling's functionResponse carried `pm.hasRelevantRules is not a function` instead of its output and the containment assertion (`response: { output: 'sibling completed' }`) failed for both `denied` and `unavailable`. No PM rules are configured in these cases, so `hasRelevantRules` returning false is the faithful stub: the sibling is then governed by its own `getDefaultPermission`, which is what the test already asserts. The assertions themselves are untouched -- this only makes the fixture callable. Also applies Prettier formatting to the new block. Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-closeout/jmun0ao3qfm --- .../src/acp-integration/session/Session.test.ts | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.test.ts b/packages/cli/src/acp-integration/session/Session.test.ts index f1810bdff5a..6c023994392 100644 --- a/packages/cli/src/acp-integration/session/Session.test.ts +++ b/packages/cli/src/acp-integration/session/Session.test.ts @@ -18491,6 +18491,11 @@ describe('Session', () => { throw new Error('policy store unavailable'); return false; }), + // No PM rules are configured here, so the independent sibling call + // is governed by its own `getDefaultPermission: 'allow'`. Without + // this stub `evaluatePermissionRules` throws on the sibling and its + // functionResponse carries the TypeError instead of its output. + hasRelevantRules: vi.fn().mockReturnValue(false), }); const bridge = { name: core.ToolNames.TOOL_CALL, @@ -18531,12 +18536,10 @@ describe('Session', () => { getDefaultPermission: vi.fn().mockResolvedValue('allow'), getDescription: () => 'independent call', toolLocations: () => [], - execute: vi - .fn() - .mockResolvedValue({ - llmContent: 'sibling completed', - returnDisplay: 'sibling completed', - }), + execute: vi.fn().mockResolvedValue({ + llmContent: 'sibling completed', + returnDisplay: 'sibling completed', + }), })), }; mockToolRegistry.getTool.mockImplementation((name: string) => From a674e3a6452248dd6ac6457960ecaa47b1400b49 Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Wed, 30 Sep 2026 16:26:23 +0800 Subject: [PATCH 13/27] fix(core): isolate deferred argument validation --- packages/core/src/tools/tool-call.test.ts | 58 +++++++++++++++++++++++ packages/core/src/tools/tool-call.ts | 19 ++++++-- 2 files changed, 73 insertions(+), 4 deletions(-) diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 0b4e993e06f..8e51e5ec79a 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -896,6 +896,64 @@ describe('ToolCallTool', () => { expect(result).toMatchObject({ arguments: { count: '3' } }); }); + it('leaves optional null placeholders for the target to normalize', async () => { + class NullTolerantTool extends MockTool { + override validateToolParams(params: { + [key: string]: unknown; + }): string | null { + return params['working_dir'] === null + ? null + : super.validateToolParams(params); + } + } + const target = new NullTolerantTool({ + name: 'agent_like', + shouldDefer: true, + params: { + type: 'object', + properties: { + prompt: { type: 'string' }, + working_dir: { type: 'string' }, + }, + required: ['prompt'], + }, + }); + const argumentsWithPlaceholder = { + prompt: 'investigate', + working_dir: null, + }; + + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: target.name, arguments: argumentsWithPlaceholder }, + ); + + expect(result).not.toHaveProperty('error'); + expect(result).toMatchObject({ arguments: argumentsWithPlaceholder }); + }); + + it('does not reserve a target schema id in the shared validator', async () => { + const target = new MockTool({ + name: 'identified_target', + shouldDefer: true, + params: { + $id: 'https://example.com/deferred-tool-precheck', + type: 'object', + properties: { prompt: { type: 'string' } }, + required: ['prompt'], + additionalProperties: false, + }, + }); + + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: target.name, arguments: { prompt: 'investigate' } }, + ); + + expect(result).not.toHaveProperty('error'); + expect(target.validateToolParams({})).toContain("'prompt'"); + }); + it('re-reads a target schema the target mutates in place after the first call', async () => { // AgentTool's refresh mutates its own parameterSchema object in place // (it adds and removes `model`/`name`), and Ajv caches a compiled diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 1ee06b092f6..17149eb99d5 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -282,9 +282,9 @@ export async function resolveDeferredToolCall( // build time; and Ajv caches a compiled schema by object identity, so a // target that mutates its own schema object in place (AgentTool's refresh // adds and removes `model`/`name`) would otherwise stay pinned to whatever - // shape it had on the first bridged call. A per-call copy resolves through - // the JSON-text tier, which still shares one compiled validator per - // distinct schema text. + // shape it had on the first bridged call. Compile the per-call copy in an + // isolated validator so it sees the current shape without reserving the + // target's `$id` in the process-shared registry. // // Only the schema layer runs here — never the target's full // validateToolParams: its value-level rules (fs stats, content scans, the @@ -318,7 +318,18 @@ export async function resolveDeferredToolCall( if (schemaClone['additionalProperties'] === false) { schemaClone['additionalProperties'] = true; } - paramsError = SchemaValidator.validate(schemaClone, argsClone); + const required = new Set( + Array.isArray(schemaClone['required']) ? schemaClone['required'] : [], + ); + for (const [name, value] of Object.entries(argsClone)) { + if (value === null && !required.has(name)) { + delete argsClone[name]; + } + } + const compiled = SchemaValidator.compileIsolated(schemaClone); + if ('validate' in compiled) { + paramsError = compiled.validate(argsClone); + } } catch { // A target whose validation throws under this pre-check must not become // a new bridge failure mode: the scheduler's build() reports the same From bacca2bc14f17f3ae5602f42e337be67d9fb15a9 Mon Sep 17 00:00:00 2001 From: yiliang114 Date: Thu, 1 Oct 2026 00:20:24 +0800 Subject: [PATCH 14/27] fix(core): relax nested additionalProperties in the bridge argument pre-check MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The surplus-key relaxation only rewrote the root keyword, so a nested `additionalProperties: false` was still compiled and enforced by the isolated validator. todo_write is exactly that family: its items schema declares `additionalProperties: false` while its own validateToolParams only type-checks the known item keys, so a bridged todo_write call carrying an extra item field was refused by a pre-check the target itself would accept — stricter than the target one level down. Relax the keyword at every level by rebuilding the schema clone with a JSON.stringify replacer (which also deep-clones), leaving nested surplus-key enforcement to the target's own build() as designed. Tests: a bridged todo_write-like target with a surplus item key now resolves with its own schema unchanged, and a nested enum violation is still refused by the pre-check. Co-authored-by: Qwen-Coder Patrol-Run: qwen-issue-patrol/jmuoaqhbp05 --- packages/core/src/tools/tool-call.test.ts | 104 ++++++++++++++++++++++ packages/core/src/tools/tool-call.ts | 17 ++-- 2 files changed, 114 insertions(+), 7 deletions(-) diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 8e51e5ec79a..855206fe407 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -856,6 +856,110 @@ describe('ToolCallTool', () => { } }); + // Same tolerance one level down: todo_write's item schema declares + // additionalProperties: false while its own validateToolParams only + // type-checks the known keys, so a nested surplus key must not trip the + // pre-check either. + const makeTodoLike = (Tool: typeof MockTool = MockTool) => + new Tool({ + name: 'todo_like', + shouldDefer: true, + params: { + type: 'object', + properties: { + todos: { + type: 'array', + items: { + type: 'object', + properties: { + id: { type: 'string' }, + content: { type: 'string' }, + status: { type: 'string', enum: ['pending', 'completed'] }, + }, + required: ['id', 'content', 'status'], + additionalProperties: false, + }, + }, + }, + required: ['todos'], + additionalProperties: false, + }, + }); + + it('leaves nested surplus-key enforcement to the target without changing its schema', async () => { + // Mutation check: relaxing only the top-level additionalProperties + // turns this red with "must NOT have additional properties". + class LenientTool extends MockTool { + override validateToolParams(): string | null { + return null; + } + } + for (const Tool of [MockTool, LenientTool]) { + const target = makeTodoLike(Tool); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { + name: target.name, + arguments: { + todos: [ + { + id: '1', + content: 'write the test', + status: 'pending', + priority: 'high', + }, + ], + }, + }, + ); + expect(result).not.toHaveProperty('error'); + expect(JSON.stringify(target.schema.parametersJsonSchema)).toContain( + '"additionalProperties":false', + ); + if ('tool' in result) { + const build = () => result.tool.build(result.arguments); + if (Tool === MockTool) { + expect(build).toThrow('must NOT have additional properties'); + } else { + expect(build).not.toThrow(); + } + } + } + }); + + it('still refuses nested schema violations when surplus keys are tolerated', async () => { + const target = makeTodoLike(); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { + name: target.name, + arguments: { + todos: [ + { + id: '1', + content: 'write the test', + status: 'bogus', + priority: 'high', + }, + ], + }, + }, + ); + + expect(result).toMatchObject({ + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: 'todo_like', + }); + if ('error' in result) { + expect( + result.error.message.startsWith(DEFERRED_TOOL_CALL_REFUSAL_PREFIX), + ).toBe(true); + expect(result.error.message).toContain( + 'must be equal to one of the allowed values', + ); + } + }); + it('still attributes wrong field types when surplus keys are present', async () => { const target = makeWebFetchLike(); const result = await resolveDeferredToolCall( diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 17149eb99d5..a9f54227c4c 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -310,14 +310,17 @@ export async function resolveDeferredToolCall( if (!options?.wasOutputTruncated && !preCheckSuppressed) { try { const argsClone = structuredClone(invocation.params.arguments); - const schemaClone = structuredClone( - target.schema.parametersJsonSchema, - ) as Record; // Some targets deliberately tolerate surplus keys (for example, Agent's - // name outside team mode). Leave that decision to their own build(). - if (schemaClone['additionalProperties'] === false) { - schemaClone['additionalProperties'] = true; - } + // name outside team mode, or todo_write's item-level extra fields). + // Leave that decision to their own build(): relax the keyword at every + // level, since a nested `additionalProperties: false` (todo_write's + // items schema) refuses bridged calls the target's own validator + // accepts. The JSON round-trip also deep-clones the schema. + const schemaClone = JSON.parse( + JSON.stringify(target.schema.parametersJsonSchema, (key, value) => + key === 'additionalProperties' && value === false ? true : value, + ), + ) as Record; const required = new Set( Array.isArray(schemaClone['required']) ? schemaClone['required'] : [], ); From 21c07c524a915b740b6abad164702667229693f5 Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Thu, 1 Oct 2026 06:32:51 +0800 Subject: [PATCH 15/27] fix(core): preserve composition branches in the bridge argument relaxation R8-1: the argument pre-check relaxed `additionalProperties: false` with a JSON reviver keyed on the property name alone, so it also rewrote the keyword inside `oneOf`/`anyOf`/`allOf`/`if`/`then`/`else`/`not` branches. In a tagged union those per-branch keywords are what distinguish the branches, so relaxing them inverts the schema's meaning instead of widening acceptance: an input that matched exactly one branch then matches two, `oneOf` fails, and the bridge refuses a call the target's own schema and build() both accept. Because the refusal carries `targetName`, the valid call also books a per-target parameter-error strike toward VALIDATION_RETRY_LOOP_THRESHOLD, so a model retrying it ends in RETRY LOOP DETECTED for a tool that works. Replace the reviver with a structural walk that relaxes the keyword only where relaxing widens acceptance. Composition branches and annotation data (`const`/`default`/`enum`/`example`/`examples`) are cloned byte-identical, and name-to-schema maps (`properties`, `$defs`, `patternProperties`, ...) are walked by value so a property literally named `additionalProperties` keeps its own schema instead of being read as the keyword. Relaxing along the ordinary `properties`/`items` path is unchanged, so the nested surplus-key tolerance this pre-check exists for still applies. The JSON round-trip still deep-clones, so the target's schema object is never mutated, and the surrounding try/catch still fails open. Adds regression tests for the tagged union (the accepted exactly-one-branch call, plus a still-refused zero-branch call so the pre-check keeps its teeth), the annotation position, and the name-as-data position. Emptying either guard set turns three of them red with the reported "params must match exactly one schema in oneOf". Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-closeout/jmuom604nil --- packages/core/src/tools/tool-call.test.ts | 129 ++++++++++++++++++++++ packages/core/src/tools/tool-call.ts | 110 ++++++++++++++++-- 2 files changed, 228 insertions(+), 11 deletions(-) diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 855206fe407..bc1c050fde3 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -960,6 +960,135 @@ describe('ToolCallTool', () => { } }); + // A deferred and hidden MCP target publishes its server's inputSchema + // unmodified, and per-branch `additionalProperties: false` is the standard + // generated tagged-union idiom: the branches are told apart BY the keyword. + const makeTaggedUnionLike = () => + new MockTool({ + name: 'mcp__srv__union', + shouldDefer: true, + params: { + type: 'object', + oneOf: [ + { + properties: { a: { type: 'string' } }, + required: ['a'], + additionalProperties: false, + }, + { + properties: { a: { type: 'string' }, b: { type: 'string' } }, + required: ['a'], + additionalProperties: false, + }, + ], + }, + }); + + it('resolves a tagged union whose branches are told apart by additionalProperties', async () => { + // R8-1: a rewrite keyed on the property name alone also flipped the + // keyword INSIDE each oneOf branch, so {a, b} matched both branches and + // oneOf (exactly one) failed. The pre-check then refused a call the + // target's own schema and build() both accept, and — because targetName + // is populated — booked a parameter-error strike toward RETRY LOOP + // DETECTED for a tool that works. Mutation check: descending into + // composition keywords turns this red with "must match exactly one + // schema in oneOf". + const target = makeTaggedUnionLike(); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: target.name, arguments: { a: 'x', b: 'y' } }, + ); + + expect(result).not.toHaveProperty('error'); + expect(result).toMatchObject({ arguments: { a: 'x', b: 'y' } }); + // Relaxing happens on a clone: the target's published schema is intact. + expect(JSON.stringify(target.schema.parametersJsonSchema)).toContain( + '"additionalProperties":false', + ); + if ('tool' in result) { + expect(() => result.tool.build(result.arguments)).not.toThrow(); + } + }); + + it('still refuses a tagged-union call that matches no branch', async () => { + // Preserving the branches is not the same as disabling the pre-check: + // `{b}` satisfies neither branch's `required: ['a']`, so it must still be + // refused and attributed to the target. + const target = makeTaggedUnionLike(); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: target.name, arguments: { b: 'y' } }, + ); + + expect(result).toMatchObject({ + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: 'mcp__srv__union', + }); + expect(result).not.toHaveProperty('tool'); + if ('error' in result) { + expect( + result.error.message.startsWith(DEFERRED_TOOL_CALL_REFUSAL_PREFIX), + ).toBe(true); + expect(result.error.message).toContain('"mcp__srv__union"'); + } + }); + + it('leaves annotation data that looks like the keyword alone', async () => { + // Same inversion one class over: `const` holds data the schema compares + // against, so rewriting the keyword inside it makes the pre-check demand + // a value the authored schema rejects. + const target = new MockTool({ + name: 'annotated_target', + shouldDefer: true, + params: { + type: 'object', + properties: { + config: { const: { additionalProperties: false } }, + }, + required: ['config'], + }, + }); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { + name: target.name, + arguments: { config: { additionalProperties: false } }, + }, + ); + + expect(result).not.toHaveProperty('error'); + }); + + it('reads a property named additionalProperties as a name, not the keyword', async () => { + // Keys of a name-to-schema map are data. The authored schema forbids this + // property outright, so the pre-check must keep refusing it rather than + // relax the prohibition away. + const target = new MockTool({ + name: 'named_property_target', + shouldDefer: true, + params: { + type: 'object', + properties: { + prompt: { type: 'string' }, + additionalProperties: false, + }, + required: ['prompt'], + }, + }); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { + name: target.name, + arguments: { prompt: 'investigate', additionalProperties: true }, + }, + ); + + expect(result).toMatchObject({ + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: 'named_property_target', + }); + }); + it('still attributes wrong field types when surplus keys are present', async () => { const target = makeWebFetchLike(); const result = await resolveDeferredToolCall( diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index a9f54227c4c..b9f996d71fb 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -87,6 +87,97 @@ function bridgeRefusal(message: string): Error { return new Error(`${DEFERRED_TOOL_CALL_REFUSAL_PREFIX}${message}`); } +/** + * Schema keywords whose subtree the relaxation below must leave byte-identical. + * Composition branches (`oneOf`/`anyOf`/`allOf`/`if`/`then`/`else`/`not`) use a + * per-branch `additionalProperties: false` to tell the branches apart, so + * relaxing it there inverts the schema's meaning instead of widening acceptance. + * Annotation keywords hold data the schema compares against, not a subschema. + */ +const VERBATIM_SCHEMA_KEYS: ReadonlySet = new Set([ + 'allOf', + 'anyOf', + 'oneOf', + 'not', + 'if', + 'then', + 'else', + 'const', + 'default', + 'enum', + 'example', + 'examples', +]); + +/** + * Schema keywords whose value maps an arbitrary NAME to a subschema. The names + * are data, so a property literally named `additionalProperties` keeps its own + * schema rather than being read as the keyword: these are walked by value only. + */ +const NAME_TO_SCHEMA_KEYS: ReadonlySet = new Set([ + '$defs', + 'definitions', + 'dependentSchemas', + 'patternProperties', + 'properties', +]); + +/** Relaxes `additionalProperties: false` in an already-cloned schema tree. */ +function relaxAdditionalPropertiesInPlace(node: unknown): void { + if (Array.isArray(node)) { + for (const item of node) { + relaxAdditionalPropertiesInPlace(item); + } + return; + } + if (!node || typeof node !== 'object') { + return; + } + const schema = node as Record; + for (const [key, value] of Object.entries(schema)) { + if (VERBATIM_SCHEMA_KEYS.has(key)) { + continue; + } + if (NAME_TO_SCHEMA_KEYS.has(key)) { + if (value && typeof value === 'object' && !Array.isArray(value)) { + const byName = value as Record; + for (const subschema of Object.values(byName)) { + relaxAdditionalPropertiesInPlace(subschema); + } + } + continue; + } + if (key === 'additionalProperties' && value === false) { + schema[key] = true; + continue; + } + relaxAdditionalPropertiesInPlace(value); + } +} + +/** + * Deep-clones a target schema with `additionalProperties: false` relaxed at + * every position where relaxing only widens acceptance. Some targets + * deliberately tolerate surplus keys (for example, Agent's `name` outside team + * mode, or todo_write's item-level extra fields), so that decision is left to + * their own build(): a nested `additionalProperties: false` (todo_write's items + * schema) must not refuse bridged calls the target's own validator accepts. + * + * The walk is structural rather than keyed on the property name alone, because + * a name-keyed rewrite also reaches the positions listed in + * `VERBATIM_SCHEMA_KEYS` and `NAME_TO_SCHEMA_KEYS`, where it makes the + * pre-check STRICTER than the schema the target publishes: an input matching + * exactly one `oneOf` branch then matches two and `oneOf` fails, and a `const` + * branch compares against rewritten data. + */ +function relaxAdditionalProperties(schema: unknown): Record { + // The JSON round-trip deep-clones, so the walk can relax in place and never + // touches the target's own schema object (which it may mutate and reuse). + const clone = JSON.parse(JSON.stringify(schema)) as Record; + relaxAdditionalPropertiesInPlace(clone); + return clone; +} + export async function resolveDeferredToolCall( registry: ToolRegistry, envelope: Record, @@ -310,17 +401,14 @@ export async function resolveDeferredToolCall( if (!options?.wasOutputTruncated && !preCheckSuppressed) { try { const argsClone = structuredClone(invocation.params.arguments); - // Some targets deliberately tolerate surplus keys (for example, Agent's - // name outside team mode, or todo_write's item-level extra fields). - // Leave that decision to their own build(): relax the keyword at every - // level, since a nested `additionalProperties: false` (todo_write's - // items schema) refuses bridged calls the target's own validator - // accepts. The JSON round-trip also deep-clones the schema. - const schemaClone = JSON.parse( - JSON.stringify(target.schema.parametersJsonSchema, (key, value) => - key === 'additionalProperties' && value === false ? true : value, - ), - ) as Record; + // Surplus-key tolerance is the target's own call, so relax the keyword + // wherever relaxing only widens acceptance — but never inside a + // composition branch or annotation data, where the rewrite inverts the + // schema's meaning and the pre-check ends up stricter than the schema the + // target publishes. See relaxAdditionalProperties. + const schemaClone = relaxAdditionalProperties( + target.schema.parametersJsonSchema, + ); const required = new Set( Array.isArray(schemaClone['required']) ? schemaClone['required'] : [], ); From df8c325776e2e9f3fd6e1d1085913e0d23431b6f Mon Sep 17 00:00:00 2001 From: "jinjing.zzj" Date: Thu, 1 Oct 2026 17:30:00 +0800 Subject: [PATCH 16/27] fix(core): close the additionalProperties relaxation at the $ref boundary MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The name-keyed exemption made composition non-lexical: $defs and definitions were walked by value, so a $ref-expressed oneOf had its branches relaxed inside the shared definition — an argument matching exactly one published branch then matched two in the pre-check's clone and the bridge refused a call the target's own schema accepts. $defs and definitions now stay byte-identical (they are reachable only through $ref, and each use site is covered directly). Two measured corrections ride along: allOf/anyOf/then/else come out of the verbatim list (relaxing under them only widens acceptance — measured 0 newly-rejected against oneOf's 1 and not's 8), and draft-07 dependencies joins the name-keyed walk so a constraint literally named additionalProperties keeps its own value instead of being inverted into always-pass. Three discriminating tests: the same tagged union written with $defs/$ref resolves and the target's own build() accepts it; allOf branches relax; a false-valued dependency named additionalProperties still refuses. Each is red under its revert (re-walking $defs, re-listing allOf, dropping dependencies from the name-keyed walk). Co-authored-by: Qwen-Coder --- packages/core/src/tools/tool-call.test.ts | 106 ++++++++++++++++++++++ packages/core/src/tools/tool-call.ts | 29 +++--- 2 files changed, 122 insertions(+), 13 deletions(-) diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 5c2e976ecdf..a67a96e1b9a 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -937,6 +937,112 @@ describe('ToolCallTool', () => { }); }); + it('resolves the same tagged union written with $defs/$ref', async () => { + // The $ref form is what generated schemas actually use. Composition is + // non-lexical through it, so the relaxation must not descend into the + // shared definitions: flipping a branch's additionalProperties there + // makes {a,b} match both and oneOf (exactly one) fails. Mutation check: + // re-adding $defs to the name-keyed walk turns this red with "must match + // exactly one schema in oneOf". + const target = new MockTool({ + name: 'mcp__srv__refunion', + shouldDefer: true, + params: { + type: 'object', + $defs: { + A: { + properties: { a: { type: 'string' } }, + required: ['a'], + additionalProperties: false, + }, + B: { + properties: { a: { type: 'string' }, b: { type: 'string' } }, + required: ['a'], + additionalProperties: false, + }, + }, + oneOf: [{ $ref: '#/$defs/A' }, { $ref: '#/$defs/B' }], + }, + }); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: target.name, arguments: { a: 'x', b: 'y' } }, + ); + + expect(result).not.toHaveProperty('error'); + expect(result).toMatchObject({ arguments: { a: 'x', b: 'y' } }); + if ('tool' in result) { + // The target's own validator (Ajv on the unmodified schema) accepts. + expect(() => result.tool.build(result.arguments)).not.toThrow(); + } + }); + + it('relaxes additionalProperties inside allOf branches, which only widens', async () => { + // allOf branches do not discriminate — both must hold — so a per-branch + // additionalProperties: false is not load-bearing the way oneOf's is. + // args {a,b} fail each strict branch (each forbids the other key) and + // pass both relaxed ones. Mutation check: re-listing allOf as verbatim + // turns this red with a refusal. + class LenientTool extends MockTool { + override validateToolParams(): string | null { + return null; + } + } + const target = new LenientTool({ + name: 'allof_target', + shouldDefer: true, + params: { + type: 'object', + allOf: [ + { + properties: { a: { type: 'string' } }, + required: ['a'], + additionalProperties: false, + }, + { + properties: { b: { type: 'string' } }, + required: ['b'], + additionalProperties: false, + }, + ], + }, + }); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: target.name, arguments: { a: 'x', b: 'y' } }, + ); + + expect(result).not.toHaveProperty('error'); + expect(result).toMatchObject({ arguments: { a: 'x', b: 'y' } }); + }); + + it('never reads a constraint literally named additionalProperties as the keyword', async () => { + // `dependencies` maps a NAME to a constraint; a dependency named + // additionalProperties with a false schema forbids that property, and + // flipping the false to true would invert it into always-pass — the + // pre-check would then resolve a call the target's own validator + // refuses. The constraint must stay false after the walk. + const target = new MockTool({ + name: 'dep_target', + shouldDefer: true, + params: { + type: 'object', + properties: { mode: { type: 'string' } }, + dependencies: { additionalProperties: false }, + }, + }); + const result = await resolveDeferredToolCall( + makeRegistry([target], new Set([target.name])), + { name: target.name, arguments: { additionalProperties: 'x' } }, + ); + + expect(result).toMatchObject({ + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: 'dep_target', + }); + expect(result).not.toHaveProperty('tool'); + }); + it('still attributes wrong field types when surplus keys are present', async () => { const target = makeWebFetchLike(); const result = await resolveDeferredToolCall( diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index b9f996d71fb..794ea0d377f 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -89,34 +89,37 @@ function bridgeRefusal(message: string): Error { /** * Schema keywords whose subtree the relaxation below must leave byte-identical. - * Composition branches (`oneOf`/`anyOf`/`allOf`/`if`/`then`/`else`/`not`) use a - * per-branch `additionalProperties: false` to tell the branches apart, so - * relaxing it there inverts the schema's meaning instead of widening acceptance. - * Annotation keywords hold data the schema compares against, not a subschema. + * `oneOf`/`not` discriminate: a per-branch `additionalProperties: false` tells + * branches apart, so relaxing it there inverts the schema's meaning instead of + * widening acceptance (`if` selects a branch by the same mechanism). Annotation + * keywords hold data the schema compares against, not a subschema. `$defs` and + * `definitions` are reached only through `$ref`: they are shared definitions, + * and relaxing inside one silently rewrites every branch that references it — + * each use site is already covered directly by the walk above. + * (`allOf`/`anyOf`/`then`/`else` are deliberately absent: relaxing under them + * only widens acceptance, so the walk descends.) */ const VERBATIM_SCHEMA_KEYS: ReadonlySet = new Set([ - 'allOf', - 'anyOf', 'oneOf', 'not', 'if', - 'then', - 'else', 'const', 'default', 'enum', 'example', 'examples', + '$defs', + 'definitions', ]); /** - * Schema keywords whose value maps an arbitrary NAME to a subschema. The names - * are data, so a property literally named `additionalProperties` keeps its own - * schema rather than being read as the keyword: these are walked by value only. + * Schema keywords whose value maps an arbitrary NAME to a subschema or + * constraint. The names are data, so a property literally named + * `additionalProperties` keeps its own schema rather than being read as the + * keyword: these are walked by value only. */ const NAME_TO_SCHEMA_KEYS: ReadonlySet = new Set([ - '$defs', - 'definitions', + 'dependencies', 'dependentSchemas', 'patternProperties', 'properties', From bd31eac296c79871b8671025e5403090ea11d0e9 Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Thu, 1 Oct 2026 19:35:21 +0800 Subject: [PATCH 17/27] fix(core): preserve bridge policy denials and media defaults --- .../src/acp-integration/session/Session.ts | 10 +++- packages/core/src/code-mode/scheduler.test.ts | 44 ++++++++++++++ packages/core/src/core/coreToolScheduler.ts | 60 ++++++++++++------- packages/core/src/tools/tool-call.test.ts | 31 ++++++++++ packages/core/src/tools/tool-call.ts | 15 +++++ 5 files changed, 138 insertions(+), 22 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.ts b/packages/cli/src/acp-integration/session/Session.ts index 5f158ae689f..602a2727a65 100644 --- a/packages/cli/src/acp-integration/session/Session.ts +++ b/packages/cli/src/acp-integration/session/Session.ts @@ -12,7 +12,10 @@ import { } from '@qwen-code/qwen-code-core/hooks/hook-execution-context.js'; import { shellResultText } from '@qwen-code/qwen-code-core/shellResult'; -import { evaluateMediaPolicyToolCall } from '@qwen-code/qwen-code-core/omni/policy/model-access.js'; +import { + evaluateMediaPolicyToolCall, + resolveMediaPolicyModelAccess, +} from '@qwen-code/qwen-code-core/omni/policy/model-access.js'; import { Buffer } from 'node:buffer'; import { randomUUID } from 'node:crypto'; @@ -13736,6 +13739,11 @@ export class Session implements SessionContext { ); } const resolution = await resolveDeferredToolCall(toolRegistry, args, { + getDefaultArgumentNames: (targetName) => + Object.keys( + resolveMediaPolicyModelAccess(this.config, targetName) + .defaultArguments, + ), // Thread the real configured depth so the ACP frontend applies the // same depth-gated AgentTool re-admission as the terminal scheduler // and tool_search — the exclusion contract must be consistent across diff --git a/packages/core/src/code-mode/scheduler.test.ts b/packages/core/src/code-mode/scheduler.test.ts index 50322560e9b..cdb0d1a30b6 100644 --- a/packages/core/src/code-mode/scheduler.test.ts +++ b/packages/core/src/code-mode/scheduler.test.ts @@ -17,6 +17,8 @@ import { makeFakeConfig } from '../test-utils/config.js'; import { MockTool } from '../test-utils/mock-tool.js'; import { ExecTool } from '../tools/exec.js'; import { ToolSearchTool } from '../tools/tool-search.js'; +import { ToolCallTool } from '../tools/tool-call.js'; +import { ToolErrorType } from '../tools/tool-error.js'; import { getToolCallRuntime } from './tool-call-runtime.js'; import { Kind, @@ -795,6 +797,48 @@ describe('CodeModeOnly scheduler dispatch', () => { ); }); + it.each([ + ['read_probe', 'model', ToolErrorType.EXECUTION_DENIED], + ['agent', 'model', ToolErrorType.INVALID_TOOL_PARAMS], + ['read_probe', 'code_mode', ToolErrorType.INVALID_TOOL_PARAMS], + ] as const)( + 'preserves the call surface policy for malformed bridged %s calls from %s', + async (name, source, errorType) => { + const execute = vi.fn(); + const { run, completed, call } = setup([ + new ToolCallTool(), + new MockTool({ name: 'tool_search' }), + new MockTool({ + name, + shouldDefer: true, + params: { + type: 'object', + properties: { value: { type: 'string' } }, + required: ['value'], + }, + execute, + }), + ]); + + for (let attempt = 0; attempt < 3; attempt++) { + completed.mockClear(); + await run(`bridge-${attempt}`, 'prompt-bridge-policy', undefined, { + name: 'tool_call', + args: { name, arguments: {} }, + source, + }); + expect(call()?.response.errorType).toBe(errorType); + if (errorType === ToolErrorType.EXECUTION_DENIED) { + expect(call()?.response.error.message).toContain('CodeModeOnly'); + expect(JSON.stringify(call()?.response)).not.toContain( + 'RETRY LOOP DETECTED', + ); + } + } + expect(execute).not.toHaveBeenCalled(); + }, + ); + it('enforces a restricted agent allowlist inside exec', async () => { const read = vi.fn().mockResolvedValue({ llmContent: 'read ok', diff --git a/packages/core/src/core/coreToolScheduler.ts b/packages/core/src/core/coreToolScheduler.ts index 15cae11f7c7..07257aa0b4f 100644 --- a/packages/core/src/core/coreToolScheduler.ts +++ b/packages/core/src/core/coreToolScheduler.ts @@ -36,7 +36,10 @@ import type { EditorType } from '../utils/editor.js'; import type { Config } from '../config/config.js'; import type { ChatRecordingService } from '../services/chatRecordingService.js'; import { createDebugLogger } from '../utils/debugLogger.js'; -import { evaluateMediaPolicyToolCall } from '../omni/policy/model-access.js'; +import { + evaluateMediaPolicyToolCall, + resolveMediaPolicyModelAccess, +} from '../omni/policy/model-access.js'; import { sanitizeToolNameForProvider } from '../utils/tool-name-utils.js'; import { compactToolResultDisplayForHistory } from '../utils/toolResultDisplayCompaction.js'; import { @@ -2805,27 +2808,42 @@ export class CoreToolScheduler { // The owner policy must win over the argument pre-check so a denied // target keeps its specific EXECUTION_DENIED. isTargetExecutionAllowed: this.isToolExecutionAllowed, - // The permission-manager gate in _schedule owns the richer denial - // (deny-rule attribution); a pm-denied target skips the argument - // pre-check so that denial — not a parameter error for a call that - // could never run — is what the model sees. - suppressArgumentPreCheck: permissionManager - ? async (targetName: string) => { - try { - return !(await permissionManager.isToolEnabled(targetName)); - } catch (error) { - // Do not let a policy lookup failure swallow the pre-check. - // On the refusal path _schedule continues ahead of the - // permission gate, so this is the lookup's only record. - debugLogger.warn( - 'Bridge pre-check policy lookup failed for', - targetName, - error, - ); - return false; - } + getDefaultArgumentNames: (targetName) => + Object.keys( + resolveMediaPolicyModelAccess(this.config, targetName) + .defaultArguments, + ), + // The Code Mode and permission-manager gates in _schedule own their + // specific denials; a blocked target skips the argument pre-check + // instead of accruing parameter errors for a call that cannot run. + suppressArgumentPreCheck: async (targetName: string) => { + if ( + this.config.getToolMode?.() === ToolMode.CodeModeOnly && + request.executionOrigin?.kind !== 'fixed_policy' && + !isCodeModeToolCallAllowed( + canonicalToolName(targetName), + request.source ?? 'model', + ) + ) { + return true; + } + if (permissionManager) { + try { + return !(await permissionManager.isToolEnabled(targetName)); + } catch (error) { + // Do not let a policy lookup failure swallow the pre-check. + // On the refusal path _schedule continues ahead of the + // permission gate, so this is the lookup's only record. + debugLogger.warn( + 'Bridge pre-check policy lookup failed for', + targetName, + error, + ); + return false; } - : undefined, + } + return false; + }, }), ); if ('error' in resolution) { diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index a67a96e1b9a..b15db1fe8bb 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -1367,6 +1367,37 @@ describe('ToolCallTool', () => { }); }); + it('allows gate-supplied defaults without changing the declaration or caller arguments', async () => { + const target = new MockMediaPolicyTool({}); + const schemaBefore = structuredClone(target.schema.parametersJsonSchema); + const registry = makeRegistry([target], new Set([target.name])); + const defaults = vi.fn(() => ['outputDir']); + for (const args of [ + { inputPath: '/tmp/in.wav' }, + { inputPath: '/tmp/in.wav', outputDir: '/caller/out' }, + ]) { + const argsBefore = structuredClone(args); + const result = await resolveDeferredToolCall( + registry, + { name: target.name, arguments: args }, + { getDefaultArgumentNames: defaults }, + ); + expect(result).toMatchObject({ tool: target, arguments: argsBefore }); + expect(args).toEqual(argsBefore); + expect(target.schema.parametersJsonSchema).toEqual(schemaBefore); + } + expect(defaults).toHaveBeenCalledWith(target.name); + const invalid = await resolveDeferredToolCall( + registry, + { name: target.name, arguments: { outputDir: {} } }, + { getDefaultArgumentNames: defaults }, + ); + expect(invalid).toMatchObject({ + errorType: ToolErrorType.INVALID_TOOL_PARAMS, + targetName: target.name, + }); + }); + it('refuses a media-policy target whose arguments miss a model-visible required field', async () => { // With no lockedArguments the projection is the native schema, so a // bridged `{}` must still be refused here — naming the target and the diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 794ea0d377f..1dc10154681 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -77,6 +77,8 @@ export interface DeferredToolCallOptions { * sees. */ suppressArgumentPreCheck?: (targetName: string) => boolean | Promise; + /** Media-policy fields supplied by the caller's downstream modelAccess gate. */ + getDefaultArgumentNames?: (targetName: string) => readonly string[]; } export const DEFERRED_TOOL_CALL_REFUSAL_PREFIX = '[tool_call bridge refused] '; @@ -391,6 +393,8 @@ export async function resolveDeferredToolCall( // rules that assume the modelAccess gate (which both frontends run AFTER // bridge resolution) has merged those arguments back in — running it here // would refuse calls the very next stage accepts. + // Defaults remain model-visible and overridable; omit their names only + // from this clone's required list because the same gate supplies them. let paramsError: string | null = null; // A truncated response yields to the caller's truncation handling: the // arguments are incomplete for transport reasons, not a schema misreading. @@ -412,6 +416,17 @@ export async function resolveDeferredToolCall( const schemaClone = relaxAdditionalProperties( target.schema.parametersJsonSchema, ); + if ( + target.mediaPolicyDescriptor?.kind === 'media_policy' && + Array.isArray(schemaClone['required']) + ) { + const defaults = new Set( + options?.getDefaultArgumentNames?.(target.name), + ); + schemaClone['required'] = schemaClone['required'].filter( + (name) => !defaults.has(name), + ); + } const required = new Set( Array.isArray(schemaClone['required']) ? schemaClone['required'] : [], ); From 38fb20a350dfa1a5aa82476d53060c7f64f2c2a0 Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Fri, 2 Oct 2026 00:14:34 +0900 Subject: [PATCH 18/27] refactor(core): label the target's own validation error instead of pre-checking bridged arguments The pre-check re-validated bridged arguments against a relaxed copy of the target schema before the target saw them. Ten review rounds kept finding third-party schema shapes ($defs unions, composition branches, media-policy defaults) where the relaxation refused a call the target accepts, and the E2E run on #12889 showed the richer error did not make the model recover. Drop the pre-check and every option it needed (suppressArgumentPreCheck, getDefaultArgumentNames, isTargetExecutionAllowed in the resolver, the ACP catch around resolution). Keep the diagnostic: when a target reached through tool_call fails its own build(), the scheduler and the ACP session name the target and point at tool_search. This only relabels a rejection the target already made, so it cannot refuse a valid call. The previous head is preserved at archive/12901-bridge-arg-precheck. --- .../acp-integration/session/Session.test.ts | 160 ---- .../src/acp-integration/session/Session.ts | 50 +- packages/core/src/code-mode/scheduler.test.ts | 44 - .../core/src/core/coreToolScheduler.test.ts | 821 ++-------------- packages/core/src/core/coreToolScheduler.ts | 165 +--- packages/core/src/index.ts | 1 + packages/core/src/tools/tool-call.test.ts | 899 +----------------- packages/core/src/tools/tool-call.ts | 230 +---- 8 files changed, 132 insertions(+), 2238 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.test.ts b/packages/cli/src/acp-integration/session/Session.test.ts index 022732df094..0cd7ff962b3 100644 --- a/packages/cli/src/acp-integration/session/Session.test.ts +++ b/packages/cli/src/acp-integration/session/Session.test.ts @@ -18562,166 +18562,6 @@ describe('Session', () => { ); }); - it.each(['denied', 'unavailable'] as const)( - 'keeps a %s target policy failure within its ACP call', - async (policyState) => { - // The ACP half of the resolution-time policy wiring: with the - // wrapper allowed but the target denied, the denial must stay with - // ACP's own L1 enablement gate and must not become a parameter - // pre-check refusal — otherwise a denied target with malformed - // arguments gets INVALID_TOOL_PARAMS plus an invalid-parameter strike - // toward the loop stop for a call that could never run. The pm is - // wired into suppressArgumentPreCheck (not isTargetExecutionAllowed, - // which is the outer owner's execution-allowlist hook that this - // frontend does not have), so the pre-check is skipped for a denied - // target and L1 supplies both the wording and the isTrustedLiveTool - // exemption. Mutation check: removing the - // `...(pm ? { suppressArgumentPreCheck } : {})` spread in - // Session.runTool turns this red (the denial becomes an - // INVALID_TOOL_PARAMS bridge refusal naming `required property 'url'` - // and invalidToolParamErrors gains an entry). - mockConfig.getApprovalMode = vi - .fn() - .mockReturnValue(ApprovalMode.YOLO); - mockConfig.getPermissionManager = vi.fn().mockReturnValue({ - isToolEnabled: vi.fn(async (name: string) => { - if (name !== 'web_fetch') return true; - if (policyState === 'unavailable') - throw new Error('policy store unavailable'); - return false; - }), - // No PM rules are configured here, so the independent sibling call - // is governed by its own `getDefaultPermission: 'allow'`. Without - // this stub `evaluatePermissionRules` throws on the sibling and its - // functionResponse carries the TypeError instead of its output. - hasRelevantRules: vi.fn().mockReturnValue(false), - }); - const bridge = { - name: core.ToolNames.TOOL_CALL, - kind: core.Kind.Other, - description: 'Deferred tool bridge', - build: vi.fn((params: Record) => ({ params })), - }; - const toolSearch = { - name: core.ToolNames.TOOL_SEARCH, - kind: core.Kind.Other, - description: 'Deferred tool discovery', - build: vi.fn((params: Record) => ({ params })), - }; - const target = { - name: 'web_fetch', - kind: core.Kind.Other, - description: 'Fetches a URL', - schema: { - parametersJsonSchema: { - type: 'object', - properties: { - url: { type: 'string' }, - prompt: { type: 'string' }, - }, - required: ['url', 'prompt'], - additionalProperties: false, - }, - }, - build: vi.fn(), - }; - const sibling = { - name: 'sibling_tool', - kind: core.Kind.Other, - displayName: 'Sibling', - description: 'An independent call', - build: vi.fn((params: Record) => ({ - params, - getDefaultPermission: vi.fn().mockResolvedValue('allow'), - getDescription: () => 'independent call', - toolLocations: () => [], - execute: vi.fn().mockResolvedValue({ - llmContent: 'sibling completed', - returnDisplay: 'sibling completed', - }), - })), - }; - mockToolRegistry.getTool.mockImplementation((name: string) => - name === bridge.name - ? bridge - : name === target.name - ? target - : name === toolSearch.name - ? toolSearch - : name === sibling.name - ? sibling - : undefined, - ); - mockToolRegistry.ensureTool.mockImplementation( - async (name: string) => - name === bridge.name - ? bridge - : name === target.name - ? target - : name === toolSearch.name - ? toolSearch - : undefined, - ); - mockToolRegistry.isDeferredAndHidden.mockImplementation( - (name: string) => name === target.name, - ); - const toolLoopState = { - totalToolCalls: 0, - invalidToolParamErrors: new Map(), - toolCallKeyCounts: new Map(), - maxToolCallKeyRepeat: 0, - loopDetected: false, - }; - - const result = await ( - session as unknown as { - runToolCalls: ( - abortSignal: AbortSignal, - promptId: string, - calls: FunctionCall[], - loopState: typeof toolLoopState, - ) => Promise<{ parts: Part[] }>; - } - ).runToolCalls( - new AbortController().signal, - 'prompt-tool-call-bridge-denied', - [ - { - id: 'bridge-denied-call', - name: core.ToolNames.TOOL_CALL, - args: { name: target.name, arguments: {} }, - }, - { id: 'independent-call', name: sibling.name, args: {} }, - ], - toolLoopState, - ); - - const errorText = String( - result.parts[0]?.functionResponse?.response?.['error'], - ); - // ACP's own L1 wording, not the bridge's scheduler-flavoured message - // that points an operator at an execution allowlist / disallowedTools - // blocklist this frontend never reads. - expect(errorText).toContain( - policyState === 'denied' - ? 'Tool "web_fetch" is disabled.' - : 'policy store unavailable', - ); - expect(errorText).not.toContain( - "not permitted by this agent's tool policy", - ); - expect(errorText).not.toContain("required property 'url'"); - expect(toolLoopState.invalidToolParamErrors.size).toBe(0); - expect(target.build).not.toHaveBeenCalled(); - expect(result.parts).toHaveLength(2); - expect(result.parts[1]?.functionResponse).toMatchObject({ - id: 'independent-call', - name: sibling.name, - response: { output: 'sibling completed' }, - }); - }, - ); - it('marks a disabled ACP tool_call as a bridge refusal', async () => { mockConfig.getPermissionManager = vi.fn().mockReturnValue({ isToolEnabled: vi.fn().mockResolvedValue(false), diff --git a/packages/cli/src/acp-integration/session/Session.ts b/packages/cli/src/acp-integration/session/Session.ts index 8395e25d9d7..76fcf1fba20 100644 --- a/packages/cli/src/acp-integration/session/Session.ts +++ b/packages/cli/src/acp-integration/session/Session.ts @@ -12,10 +12,7 @@ import { } from '@qwen-code/qwen-code-core/hooks/hook-execution-context.js'; import { shellResultText } from '@qwen-code/qwen-code-core/shellResult'; -import { - evaluateMediaPolicyToolCall, - resolveMediaPolicyModelAccess, -} from '@qwen-code/qwen-code-core/omni/policy/model-access.js'; +import { evaluateMediaPolicyToolCall } from '@qwen-code/qwen-code-core/omni/policy/model-access.js'; import { Buffer } from 'node:buffer'; import { randomUUID } from 'node:crypto'; @@ -115,6 +112,7 @@ import { ToolErrorType, DEFERRED_TOOL_CALL_REFUSAL_PREFIX, DEFERRED_TOOL_CALL_CANCELLATION_PREFIX, + describeBridgedArgumentError, resolveDeferredToolCall, CreateSubSessionTool, fireNotificationHook, @@ -13699,6 +13697,7 @@ export class Session implements SessionContext { } let toolName = fc.name; + let bridgedThroughToolCall = false; if ( !appExecution && this.config.getToolMode?.() === ToolMode.CodeModeOnly && @@ -13751,43 +13750,13 @@ export class Session implements SessionContext { ); } const resolution = await resolveDeferredToolCall(toolRegistry, args, { - getDefaultArgumentNames: (targetName) => - Object.keys( - resolveMediaPolicyModelAccess(this.config, targetName) - .defaultArguments, - ), // Thread the real configured depth so the ACP frontend applies the // same depth-gated AgentTool re-admission as the terminal scheduler // and tool_search — the exclusion contract must be consistent across // all three frontends (wenshao triage follow-up). Omitting it would // fail closed, not open, but the corner case should agree everywhere. maxSubagentDepth: this.config.getMaxSubagentDepth(), - // Mirror the scheduler's split of the two bridge policy options: - // isTargetExecutionAllowed is the OUTER OWNER's execution allowlist - // (only agent-core supplies one — this frontend has no such knob), - // while the permission manager owns enablement / deny rules. So the pm - // goes into suppressArgumentPreCheck: a pm-denied target skips the - // argument pre-check (no INVALID_TOOL_PARAMS strike for a call that - // could never run) and the L1 enablement gate below stays ACP's only - // policy authority, keeping its own denial wording and its - // isTrustedLiveTool exemption. Wiring the pm into - // isTargetExecutionAllowed instead made resolution short-circuit with - // the bridge's scheduler-flavoured message, which names an execution - // allowlist and a disallowedTools blocklist that ACP never reads. - ...(pm - ? { - suppressArgumentPreCheck: async (targetName: string) => - !(await pm.isToolEnabled(targetName)), - } - : {}), - }).catch( - ( - error: unknown, - ): Awaited> => ({ - error: error instanceof Error ? error : new Error(String(error)), - errorType: ToolErrorType.UNHANDLED_EXCEPTION, - }), - ); + }); const bridgeCancellation = cancelBeforeExecutionIfAborted(toolName); if (bridgeCancellation) return bridgeCancellation; if ('error' in resolution) { @@ -13803,6 +13772,7 @@ export class Session implements SessionContext { toolName = resolution.tool.name; args = resolution.arguments; tool = resolution.tool; + bridgedThroughToolCall = true; } if (!tool) { @@ -15977,7 +15947,15 @@ export class Session implements SessionContext { : undefined, }; } catch (e) { - const error = e instanceof Error ? e : new Error(String(e)); + const caught = e instanceof Error ? e : new Error(String(e)); + // Same labelling as the scheduler: a target reached through + // tool_call names itself when its own build() rejects the arguments. + const error = + bridgedThroughToolCall && !toolBuildSucceeded + ? new Error( + describeBridgedArgumentError(toolName, caught.message), + ) + : caught; const hooksEnabledForError = !this.config.getDisableAllHooks?.(); const messageBusForError = this.config.getMessageBus?.(); const executionTimeoutException = diff --git a/packages/core/src/code-mode/scheduler.test.ts b/packages/core/src/code-mode/scheduler.test.ts index cdb0d1a30b6..50322560e9b 100644 --- a/packages/core/src/code-mode/scheduler.test.ts +++ b/packages/core/src/code-mode/scheduler.test.ts @@ -17,8 +17,6 @@ import { makeFakeConfig } from '../test-utils/config.js'; import { MockTool } from '../test-utils/mock-tool.js'; import { ExecTool } from '../tools/exec.js'; import { ToolSearchTool } from '../tools/tool-search.js'; -import { ToolCallTool } from '../tools/tool-call.js'; -import { ToolErrorType } from '../tools/tool-error.js'; import { getToolCallRuntime } from './tool-call-runtime.js'; import { Kind, @@ -797,48 +795,6 @@ describe('CodeModeOnly scheduler dispatch', () => { ); }); - it.each([ - ['read_probe', 'model', ToolErrorType.EXECUTION_DENIED], - ['agent', 'model', ToolErrorType.INVALID_TOOL_PARAMS], - ['read_probe', 'code_mode', ToolErrorType.INVALID_TOOL_PARAMS], - ] as const)( - 'preserves the call surface policy for malformed bridged %s calls from %s', - async (name, source, errorType) => { - const execute = vi.fn(); - const { run, completed, call } = setup([ - new ToolCallTool(), - new MockTool({ name: 'tool_search' }), - new MockTool({ - name, - shouldDefer: true, - params: { - type: 'object', - properties: { value: { type: 'string' } }, - required: ['value'], - }, - execute, - }), - ]); - - for (let attempt = 0; attempt < 3; attempt++) { - completed.mockClear(); - await run(`bridge-${attempt}`, 'prompt-bridge-policy', undefined, { - name: 'tool_call', - args: { name, arguments: {} }, - source, - }); - expect(call()?.response.errorType).toBe(errorType); - if (errorType === ToolErrorType.EXECUTION_DENIED) { - expect(call()?.response.error.message).toContain('CodeModeOnly'); - expect(JSON.stringify(call()?.response)).not.toContain( - 'RETRY LOOP DETECTED', - ); - } - } - expect(execute).not.toHaveBeenCalled(); - }, - ); - it('enforces a restricted agent allowlist inside exec', async () => { const read = vi.fn().mockResolvedValue({ llmContent: 'read ok', diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index 7b3fbad44b7..7f9b8711022 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -1377,6 +1377,12 @@ describe('CoreToolScheduler', () => { }, ); + const URL_REQUIRED_PARAMS = { + type: 'object', + properties: { url: { type: 'string' } }, + required: ['url'], + }; + /** tool_call bridge + hidden deferred MockTool (mcp__github__create_issue). */ function bridgeWithDeferred( deferredOptions: Partial[0]> = {}, @@ -1639,666 +1645,6 @@ describe('CoreToolScheduler', () => { expect(functionResponseOf(completed)?.name).toBe(ToolNames.TOOL_CALL); }); - it('denies a policy-blocked bridged target before validating its arguments', async () => { - // The owner execution allowlist must win over the bridge argument - // pre-check: a denied target gets the specific EXECUTION_DENIED naming - // the policy, not an INVALID_TOOL_PARAMS parameter error for a call that - // could never run (and the denied tool's validator never executes). The - // allowlist test above does not discriminate — its MockTool carries no - // required schema, so the pre-check passes it. Mutation check: running - // the pre-check ahead of the policy gate turns this red with "params - // must have required property 'url'". - const execute = vi.fn(); - const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); - const deferred = new MockTool({ - name: 'web_fetch', - shouldDefer: true, - params: { - type: 'object', - properties: { - url: { type: 'string' }, - prompt: { type: 'string' }, - }, - required: ['url', 'prompt'], - additionalProperties: false, - }, - execute, - }); - const { scheduler, onAllToolCallsComplete } = - createSchedulerForLegacyToolTests({ - toolsByName: new Map([ - [bridge.name, bridge], - [deferred.name, deferred], - ]), - deferredHiddenNames: new Set([deferred.name]), - isToolExecutionAllowed: (name: string) => name !== 'web_fetch', - }); - - await scheduler.schedule( - { - callId: 'bridge-deny-before-precheck', - name: ToolNames.TOOL_CALL, - args: { name: deferred.name, arguments: {} }, - isClientInitiated: false, - prompt_id: 'prompt-bridge-deny-before-precheck', - }, - new AbortController().signal, - ); - - expect(execute).not.toHaveBeenCalled(); - const completed = onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.errorType).toBe(ToolErrorType.EXECUTION_DENIED); - expect(completed.response.error?.message).toContain( - "is not permitted by this agent's tool policy", - ); - expect(completed.response.error?.message).not.toContain( - "required property 'url'", - ); - } - }); - - it('denies a permission-manager-blocked bridged target with the loop denial, not a parameter pre-check refusal', async () => { - // The scheduler applies two execution policies to a bridged target: the - // owner allowlist (forwarded into resolution, tested above) and the - // permission-manager enablement gate in _schedule. A pm-denied target - // with schema-invalid arguments must still surface the loop's own - // denial — with its deny-rule attribution — rather than the bridge's - // INVALID_TOOL_PARAMS pre-check refusal, which would tell the model to - // fix arguments on a call that could never run. Mutation check: - // dropping the suppressArgumentPreCheck wiring in - // resolveToolCallBridgeRequest turns this red with "params must have - // required property 'url'". - const execute = vi.fn(); - const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); - const deferred = new MockTool({ - name: 'web_fetch', - shouldDefer: true, - params: { - type: 'object', - properties: { - url: { type: 'string' }, - prompt: { type: 'string' }, - }, - required: ['url', 'prompt'], - additionalProperties: false, - }, - execute, - }); - const { scheduler, onAllToolCallsComplete } = - createSchedulerForLegacyToolTests({ - toolsByName: new Map([ - [bridge.name, bridge], - [deferred.name, deferred], - ]), - deferredHiddenNames: new Set([deferred.name]), - permissionManager: { - isToolEnabled: async (name: string) => name !== 'web_fetch', - findMatchingDenyRule: () => 'permissions.deny: web_fetch', - }, - }); - - await scheduler.schedule( - { - callId: 'bridge-pm-deny-target', - name: ToolNames.TOOL_CALL, - args: { name: deferred.name, arguments: {} }, - isClientInitiated: false, - prompt_id: 'prompt-bridge-pm-deny-target', - }, - new AbortController().signal, - ); - - expect(execute).not.toHaveBeenCalled(); - const completed = onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.errorType).toBe(ToolErrorType.EXECUTION_DENIED); - expect(completed.response.error?.message).toContain( - 'permissions.deny: web_fetch', - ); - expect(completed.response.error?.message).not.toContain( - "required property 'url'", - ); - } - }); - - it('logs the policy lookup failure and still pre-checks a bridged target', async () => { - // The try/catch around the permission-manager lookup is the only thing - // keeping a policy-store rejection out of _schedule's Promise.all, whose - // try closes with a bare finally: an uncaught rejection there rejects the - // public schedule() and fails the whole batch instead of one call. The - // guard returns false ("do not suppress"), so the pre-check still runs and - // still names the missing field. The swallow is logged because _schedule - // pushes the bridge refusal and continues ahead of the permission gate, so - // nothing else records that the lookup failed. Mutation check: inlining - // !(await permissionManager.isToolEnabled(...)) without the try turns this - // red — schedule() rejects, or the call completes EXECUTION_DENIED instead - // of INVALID_TOOL_PARAMS. Deleting only the debugLogger.warn turns the log - // assertion red. - const execute = vi.fn(); - const policyError = new Error('policy store unavailable'); - const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); - const deferred = new MockTool({ - name: 'web_fetch', - shouldDefer: true, - params: { - type: 'object', - properties: { - url: { type: 'string' }, - prompt: { type: 'string' }, - }, - required: ['url', 'prompt'], - additionalProperties: false, - }, - execute, - }); - const { scheduler, onAllToolCallsComplete } = - createSchedulerForLegacyToolTests({ - toolsByName: new Map([ - [bridge.name, bridge], - [deferred.name, deferred], - ]), - deferredHiddenNames: new Set([deferred.name]), - permissionManager: { - isToolEnabled: async (name: string) => { - if (name === 'web_fetch') { - throw policyError; - } - return true; - }, - findMatchingDenyRule: () => undefined, - }, - }); - debugLoggerWarnSpy.mockClear(); - - await scheduler.schedule( - { - callId: 'bridge-policy-lookup-throws', - name: ToolNames.TOOL_CALL, - args: { name: deferred.name, arguments: {} }, - isClientInitiated: false, - prompt_id: 'prompt-bridge-policy-lookup-throws', - }, - new AbortController().signal, - ); - - expect(execute).not.toHaveBeenCalled(); - expect(debugLoggerWarnSpy).toHaveBeenCalledWith( - expect.stringContaining('Bridge pre-check policy lookup failed for'), - deferred.name, - policyError, - ); - const completed = onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.errorType).toBe( - ToolErrorType.INVALID_TOOL_PARAMS, - ); - expect(completed.response.error?.message).toContain( - "required property 'url'", - ); - } - }); - - it('accrues bridge argument refusals per target for retry-loop detection', async () => { - // The refusal carries the validated targetName for exactly this - // accounting: alternating broken bridged calls against two distinct - // targets must strike per-target counters. Keyed on the wrapper - // (`tool_call`) instead, each strike prunes the other target's counter - // and neither ever reaches VALIDATION_RETRY_LOOP_THRESHOLD — the loop - // this PR exists to stop escapes the early stop directive. Mutation - // check: keying the refusal branch on reqInfo.name turns this red. - const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); - const makeDeferred = (name: string, required: string[]) => - new MockTool({ - name, - shouldDefer: true, - params: { - type: 'object', - properties: Object.fromEntries( - required.map((key) => [key, { type: 'string' }]), - ), - required, - additionalProperties: false, - }, - }); - const writeFile = makeDeferred('write_file', ['file_path', 'content']); - const webFetch = makeDeferred('web_fetch', ['url', 'prompt']); - const { scheduler, onAllToolCallsComplete } = - createSchedulerForLegacyToolTests({ - toolsByName: new Map([ - [bridge.name, bridge], - [writeFile.name, writeFile], - [webFetch.name, webFetch], - ]), - deferredHiddenNames: new Set([writeFile.name, webFetch.name]), - }); - - const scheduleAlternatingBatch = async (batchId: number) => { - onAllToolCallsComplete.mockClear(); - await scheduler.schedule( - [ - { - callId: `bridge-loop-${batchId}-write`, - name: ToolNames.TOOL_CALL, - args: { name: 'write_file', arguments: {} }, - isClientInitiated: false, - prompt_id: 'prompt-bridge-loop', - }, - { - callId: `bridge-loop-${batchId}-fetch`, - name: ToolNames.TOOL_CALL, - args: { name: 'web_fetch', arguments: {} }, - isClientInitiated: false, - prompt_id: 'prompt-bridge-loop', - }, - ], - new AbortController().signal, - ); - await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); - return onAllToolCallsComplete.mock.calls[0][0] as ToolCall[]; - }; - - for (const batch of [ - await scheduleAlternatingBatch(1), - await scheduleAlternatingBatch(2), - ]) { - expect(batch).toHaveLength(2); - for (const completed of batch) { - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.errorType).toBe( - ToolErrorType.INVALID_TOOL_PARAMS, - ); - expect(completed.response.error?.message).not.toContain( - 'RETRY LOOP DETECTED', - ); - } - } - } - - const third = await scheduleAlternatingBatch(3); - expect(third).toHaveLength(2); - for (const completed of third) { - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.errorType).toBe( - ToolErrorType.INVALID_TOOL_PARAMS, - ); - expect(completed.response.error?.message).toContain( - 'RETRY LOOP DETECTED', - ); - } - } - }); - - it('does not preserve stale bridge-refusal counters across a policy-denied bridge batch', async () => { - // An EXECUTION_DENIED bridge refusal accrues nothing, so it must not - // keep the denied target's stale retry counters alive through the - // batch-start prune — otherwise the next bridged malformed call fires - // RETRY LOOP DETECTED one failure early. Mutation check: keying the - // presence-set widening on targetName presence alone (instead of the - // INVALID_TOOL_PARAMS error type) turns this red. - const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); - const writeFile = new MockTool({ - name: 'write_file', - shouldDefer: true, - params: { - type: 'object', - properties: { - file_path: { type: 'string' }, - content: { type: 'string' }, - }, - required: ['file_path', 'content'], - additionalProperties: false, - }, - }); - let policyDenies = false; - const { scheduler, onAllToolCallsComplete } = - createSchedulerForLegacyToolTests({ - toolsByName: new Map([ - [bridge.name, bridge], - [writeFile.name, writeFile], - ]), - deferredHiddenNames: new Set([writeFile.name]), - isToolExecutionAllowed: () => !policyDenies, - }); - - const runBatch = async ( - batchId: string, - requests: Array<{ - name: string; - args: Record; - }>, - ) => { - onAllToolCallsComplete.mockClear(); - await scheduler.schedule( - requests.map((request, index) => ({ - callId: `${batchId}-${index}`, - ...request, - isClientInitiated: false, - prompt_id: 'prompt-bridge-stale-counter', - })), - new AbortController().signal, - ); - await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); - return onAllToolCallsComplete.mock.calls[0][0] as ToolCall[]; - }; - - // Seed the target's bridge-channel counter to threshold - 1. - for (const batch of ['seed-1', 'seed-2']) { - const [completed] = await runBatch(batch, [ - { - name: ToolNames.TOOL_CALL, - args: { name: 'write_file', arguments: {} }, - }, - ]); - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.errorType).toBe( - ToolErrorType.INVALID_TOOL_PARAMS, - ); - expect(completed.response.error?.message).not.toContain( - 'RETRY LOOP DETECTED', - ); - } - } - - // A batch whose only request is a policy-denied bridged write_file: - // records nothing, and must not retain the stale counter. - policyDenies = true; - const [denied] = await runBatch('denied', [ - { - name: ToolNames.TOOL_CALL, - args: { name: 'write_file', arguments: {} }, - }, - ]); - expect(denied.status).toBe('error'); - if (denied.status === 'error') { - expect(denied.response.errorType).toBe(ToolErrorType.EXECUTION_DENIED); - } - policyDenies = false; - - // The next bridged malformed call restarts at 1, not threshold. - const [after] = await runBatch('after', [ - { - name: ToolNames.TOOL_CALL, - args: { name: 'write_file', arguments: {} }, - }, - ]); - expect(after.status).toBe('error'); - if (after.status === 'error') { - expect(after.response.errorType).toBe(ToolErrorType.INVALID_TOOL_PARAMS); - expect(after.response.error?.message).not.toContain( - 'RETRY LOOP DETECTED', - ); - } - }); - - it('accrues bridged pre-check refusals independently from the same target’s direct failures', async () => { - // A bridged pre-check refusal and a direct validation failure of the - // SAME target must not share a retry namespace: recordRetryableToolError - // prunes same-prefix keys on every record, so an unmarked shared key - // would let the two channels reset each other every batch and the mixed - // loop would never reach the threshold — the loop this PR exists to - // stop. Mutation check: keying the refusal branch on the bare - // targetName (no channel marker) turns this red. - const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); - const writeFile = new MockTool({ - name: 'write_file', - shouldDefer: true, - params: { - type: 'object', - properties: { - file_path: { type: 'string' }, - content: { type: 'string' }, - }, - required: ['file_path', 'content'], - additionalProperties: false, - }, - }); - const { scheduler, onAllToolCallsComplete } = - createSchedulerForLegacyToolTests({ - toolsByName: new Map([ - [bridge.name, bridge], - [writeFile.name, writeFile], - ]), - deferredHiddenNames: new Set([writeFile.name]), - }); - - const runMixedBatch = async (batchId: number) => { - onAllToolCallsComplete.mockClear(); - await scheduler.schedule( - [ - { - callId: `mixed-${batchId}-bridge`, - name: ToolNames.TOOL_CALL, - args: { name: 'write_file', arguments: {} }, - isClientInitiated: false, - prompt_id: 'prompt-bridge-mixed-channel', - }, - { - callId: `mixed-${batchId}-direct`, - name: 'write_file', - args: {}, - isClientInitiated: false, - prompt_id: 'prompt-bridge-mixed-channel', - }, - ], - new AbortController().signal, - ); - await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); - return onAllToolCallsComplete.mock.calls[0][0] as ToolCall[]; - }; - - for (const batch of [await runMixedBatch(1), await runMixedBatch(2)]) { - expect(batch).toHaveLength(2); - for (const completed of batch) { - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.errorType).toBe( - ToolErrorType.INVALID_TOOL_PARAMS, - ); - expect(completed.response.error?.message).not.toContain( - 'RETRY LOOP DETECTED', - ); - } - } - } - - const third = await runMixedBatch(3); - expect(third).toHaveLength(2); - const [bridgeRefusal, directFailure] = third; - expect(bridgeRefusal.status).toBe('error'); - expect(directFailure.status).toBe('error'); - // Both halves of the claim this test's name makes: each channel reaches - // the threshold on its own key by the third batch. - if (bridgeRefusal.status === 'error') { - expect(bridgeRefusal.response.error?.message).toContain( - 'RETRY LOOP DETECTED', - ); - } - if (directFailure.status === 'error') { - expect(directFailure.response.error?.message).toContain( - 'RETRY LOOP DETECTED', - ); - } - }); - - it('accrues alternating bridged and direct failures of one target across separate batches', async () => { - // The channel marker alone is not enough: the batch-start presence prune - // runs per batch, so if each batch preserves only the channel it contains, - // a model that alternates between bridging a deferred target and calling - // it directly deletes the OTHER channel's counter every turn. Neither - // counter then exceeds 1, RETRY LOOP DETECTED is never injected, and the - // mixed-channel loop runs indefinitely — the stagnation this PR exists to - // stop. Mutation check: dropping `bridgeRetryToolName(r.name)` from the - // presence set in _schedule turns this red (no batch ever fires). - const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); - const writeFile = new MockTool({ - name: 'write_file', - shouldDefer: true, - params: { - type: 'object', - properties: { - file_path: { type: 'string' }, - content: { type: 'string' }, - }, - required: ['file_path', 'content'], - additionalProperties: false, - }, - }); - const { scheduler, onAllToolCallsComplete } = - createSchedulerForLegacyToolTests({ - toolsByName: new Map([ - [bridge.name, bridge], - [writeFile.name, writeFile], - ]), - deferredHiddenNames: new Set([writeFile.name]), - }); - - // One request per batch, so each batch's presence set holds exactly one - // channel plus whatever the widening adds for it. - const runSingleBatch = async ( - callId: string, - request: { name: string; args: Record }, - ) => { - onAllToolCallsComplete.mockClear(); - await scheduler.schedule( - { - callId, - name: request.name, - args: request.args, - isClientInitiated: false, - prompt_id: 'prompt-bridge-alternating', - }, - new AbortController().signal, - ); - await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); - return onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; - }; - const bridged = { - name: ToolNames.TOOL_CALL, - args: { name: 'write_file', arguments: {} }, - }; - const direct = { name: 'write_file', args: {} }; - - const messages: string[] = []; - for (const batchId of [1, 2, 3, 4, 5, 6]) { - const completed = await runSingleBatch( - `alternating-${batchId}`, - batchId % 2 === 1 ? bridged : direct, - ); - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.errorType).toBe( - ToolErrorType.INVALID_TOOL_PARAMS, - ); - messages.push(completed.response.error?.message ?? ''); - } - } - - // The bridged channel climbs 1, 2, 3 across the odd batches while the - // even batches restart the direct channel, so the directive fires exactly - // once — on the fifth batch — and never prematurely. - expect(messages).toHaveLength(6); - for (const early of messages.slice(0, 4)) { - expect(early).not.toContain('RETRY LOOP DETECTED'); - } - expect(messages[4]).toContain('RETRY LOOP DETECTED'); - expect(messages[5]).not.toContain('RETRY LOOP DETECTED'); - }); - - it('clears a target’s bridge-marked counter when a bridged call to it succeeds', async () => { - // The clearing half of the presence widening. A resolved bridge renames - // the request to the TARGET, so the batch-start presence set now keeps - // both of that target's channels and the prune no longer drops the stale - // bridge-marked count. clearRetryCountsForTool must cover the marked - // channel too, or the surviving count of 2 plus two later refusals would - // fire RETRY LOOP DETECTED prematurely. Mutation check: reverting - // clearRetryCountsForTool to the bare `${toolName}:` prefix turns this red. - const execute = vi.fn().mockResolvedValue({ - llmContent: [{ text: 'published' }], - returnDisplay: 'published', - }); - const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); - // A neutral target: PATH_ARG_KEYS (file_path/path/...) are rewritten on - // request.args during execution, which is not what this case measures. - const publishNote = new MockTool({ - name: 'publish_note', - shouldDefer: true, - execute, - params: { - type: 'object', - properties: { - note_id: { type: 'string' }, - body: { type: 'string' }, - }, - required: ['note_id', 'body'], - additionalProperties: false, - }, - }); - const { scheduler, onAllToolCallsComplete } = - createSchedulerForLegacyToolTests({ - toolsByName: new Map([ - [bridge.name, bridge], - [publishNote.name, publishNote], - ]), - deferredHiddenNames: new Set([publishNote.name]), - }); - - const runBridged = async ( - callId: string, - args: Record, - ) => { - onAllToolCallsComplete.mockClear(); - await scheduler.schedule( - { - callId, - name: ToolNames.TOOL_CALL, - args: { name: 'publish_note', arguments: args }, - isClientInitiated: false, - prompt_id: 'prompt-bridge-clear', - }, - new AbortController().signal, - ); - await vi.waitFor(() => expect(onAllToolCallsComplete).toHaveBeenCalled()); - return onAllToolCallsComplete.mock.calls[0][0][0] as ToolCall; - }; - - // Two bridged refusals take the marked channel to 2. - for (const batchId of [1, 2]) { - const completed = await runBridged(`clear-${batchId}`, {}); - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.error?.message).not.toContain( - 'RETRY LOOP DETECTED', - ); - } - } - - // A successful bridged execution of the SAME target clears both channels. - const succeeded = await runBridged('clear-success', { - note_id: 'n-1', - body: 'ok', - }); - expect(succeeded.status).toBe('success'); - expect(execute).toHaveBeenCalledTimes(1); - - // Two more refusals restart at 1 instead of inheriting the stale count. - for (const batchId of [3, 4]) { - const completed = await runBridged(`clear-after-${batchId}`, {}); - expect(completed.status).toBe('error'); - if (completed.status === 'error') { - expect(completed.response.error?.message).not.toContain( - 'RETRY LOOP DETECTED', - ); - } - } - }); - it('applies the retry-loop directive to repeated invalid tool_call envelopes', async () => { const { scheduler, onAllToolCallsComplete } = createSchedulerForLegacyToolTests({ @@ -2334,18 +1680,50 @@ describe('CoreToolScheduler', () => { expect(functionResponseOf(third)?.name).toBe(ToolNames.TOOL_CALL); }); + it('names the target when a bridged call fails its own validation (#12889)', async () => { + const { completed, deferred } = await runBridgeCall('bridge-invalid-args', { + params: URL_REQUIRED_PARAMS, + }); + + expectStatus(completed, 'error'); + expect(completed.response.errorType).toBe( + ToolErrorType.INVALID_TOOL_PARAMS, + ); + expect(functionResponseOf(completed)?.name).toBe(ToolNames.TOOL_CALL); + const message = completed.response.error?.message ?? ''; + expect(message).toContain(`Deferred tool "${deferred.name}"`); + expect(message).toContain("must have required property 'url'"); + expect(message).toContain(ToolNames.TOOL_SEARCH); + }); + + it("leaves a direct call's validation error unlabelled", async () => { + const direct = new MockTool({ + name: 'needs_url', + params: URL_REQUIRED_PARAMS, + }); + const { scheduler, onAllToolCallsComplete } = + createSchedulerForLegacyToolTests({ toolsByName: toolMap(direct) }); + + await scheduler.schedule( + toolRequest('direct-invalid-args', direct.name, {}, 'prompt-direct'), + new AbortController().signal, + ); + + const completed = firstBatch(onAllToolCallsComplete)[0]; + expectStatus(completed, 'error'); + const message = completed.response.error?.message ?? ''; + expect(message).toContain("must have required property 'url'"); + expect(message).not.toContain('Deferred tool'); + }); + it('prunes the bridge-keyed retry counter across a successful bridged execution', async () => { - // R1-18: invalid envelopes record under the bridge channel - // (`tool_call(via tool_call):`), while a successfully resolved - // envelope renames the request to the resolved TARGET before the - // batch-start prune runs — so the prune is the only mechanism that - // clears a stale bridge-channel count across a successful bridged - // execution. Interleave one: without the prune (e.g. a refactor keying - // presence by model-facing name), the count - // of 2 would survive the successful call and the next two identical - // failures would reach the threshold and inject RETRY LOOP DETECTED - // prematurely — while the direct-tool isolation test stays green, because - // there recording and prune names never diverge. + // R1-18: invalid envelopes record under the model-facing name + // (`tool_call:`), but a resolved envelope is renamed to the TARGET + // before the batch-start prune runs, so the prune alone clears a stale + // `tool_call:` count across a successful bridged execution. Without it + // (e.g. presence keyed by model-facing name) the count of 2 survives and + // the next two identical failures inject RETRY LOOP DETECTED prematurely, + // while the direct-tool isolation test (whose names never diverge) passes. const execute = vi.fn().mockResolvedValue({ llmContent: [{ text: 'issue created' }], returnDisplay: 'issue created', @@ -7780,42 +7158,21 @@ describe('CoreToolScheduler request queueing', () => { }); describe('CoreToolScheduler truncated output protection', () => { - /** AUTO_EDIT scheduler whose registry lists and resolves `tool` (plus any `extraTools`). */ - function createTruncationTestScheduler( - tool: AnyDeclarativeTool, - toolNames: string[] = [tool.name], - options: { - extraTools?: AnyDeclarativeTool[]; - deferredHiddenNames?: ReadonlySet; - } = {}, - ) { - const toolsByName = new Map([ - [tool.name, tool], - ...(options.extraTools ?? []).map( - (extra) => [extra.name, extra] as const, - ), - ]); + /** AUTO_EDIT scheduler whose registry lists and resolves only `tool`. */ + function createTruncationTestScheduler(tool: AnyDeclarativeTool) { return schedulerWithCallbacks( makeSchedulerConfig( { - getTool: (name: string) => toolsByName.get(name) ?? tool, - ensureTool: async (name: string) => toolsByName.get(name) ?? tool, - getAllToolNames: () => [ - ...toolNames, - ...(options.extraTools ?? []).map((extra) => extra.name), - ], + getTool: () => tool, + ensureTool: async () => tool, + getAllToolNames: () => [tool.name], getFunctionDeclarations: () => [], tools: new Map(), - isDeferredAndHidden: (name: string) => - options.deferredHiddenNames?.has(name) ?? false, } as unknown as ToolRegistry, { getApprovalMode: () => ApprovalMode.AUTO_EDIT, getPermissionsDeny: () => undefined, isInteractive: () => true, - // The bridged path resolves deferred tool calls, which reads the - // depth policy; makeSchedulerConfig's defaults omit it. - getMaxSubagentDepth: () => DEFAULT_MAX_SUBAGENT_DEPTH, }, ), ); @@ -7916,76 +7273,6 @@ describe('CoreToolScheduler truncated output protection', () => { ); }); - it('should prefer truncation handling over the bridge argument pre-check for a bridged write_file call', async () => { - // Same as the direct-call case above, but reached through the tool_call - // bridge (#12889): the bridge pre-check runs ahead of the truncation - // guards below, so it must yield when the request is stamped - // wasOutputTruncated — otherwise a max_tokens-cut envelope surfaces as a - // schema mismatch and the model re-sends the same oversized write. - // Mutation check: dropping the wasOutputTruncated condition from the - // pre-check turns this red with "params must have required property - // 'content'". - const writeFileConfig = { - getProjectRoot: () => '/tmp', - getTargetDir: () => '/tmp', - getFileSystemService: () => ({ - readTextFile: vi.fn(), - writeTextFile: vi.fn(), - }), - getDefaultFileEncoding: () => undefined, - setApprovalMode: vi.fn(), - } as unknown as Config; - const writeFileTool = new WriteFileTool(writeFileConfig); - const bridge = new MockTool({ name: ToolNames.TOOL_CALL }); - const toolSearch = new MockTool({ name: ToolNames.TOOL_SEARCH }); - const { scheduler, onAllToolCallsComplete } = createTruncationTestScheduler( - writeFileTool, - [WriteFileTool.Name], - { - extraTools: [bridge, toolSearch], - deferredHiddenNames: new Set([WriteFileTool.Name]), - }, - ); - - await scheduler.schedule( - [ - { - callId: '1', - name: ToolNames.TOOL_CALL, - args: { - name: WriteFileTool.Name, - arguments: { file_path: '/tmp/test.txt' }, - }, - isClientInitiated: false, - prompt_id: 'prompt-id-bridge-write-file-truncated', - wasOutputTruncated: true, - }, - ], - new AbortController().signal, - ); - - await vi.waitFor(() => { - expect(onAllToolCallsComplete).toHaveBeenCalled(); - }); - - const completedCalls = onAllToolCallsComplete.mock - .calls[0][0] as ToolCall[]; - expect(completedCalls).toHaveLength(1); - const completedCall = completedCalls[0]; - expect(completedCall.status).toBe('error'); - - if (completedCall.status === 'error') { - const errorMessage = completedCall.response.error?.message; - expect(errorMessage).toContain('truncated due to max_tokens limit'); - expect(errorMessage).toContain( - 'rejected to prevent writing truncated content', - ); - expect(errorMessage).not.toContain( - "params must have required property 'content'", - ); - } - }); - it('should inject retry loop directive after repeated truncated write_file rejections', async () => { const { scheduler, onAllToolCallsComplete } = createTruncationTestScheduler(writeFileTool()); diff --git a/packages/core/src/core/coreToolScheduler.ts b/packages/core/src/core/coreToolScheduler.ts index b78edb7a675..527cd831f3a 100644 --- a/packages/core/src/core/coreToolScheduler.ts +++ b/packages/core/src/core/coreToolScheduler.ts @@ -36,10 +36,7 @@ import type { EditorType } from '../utils/editor.js'; import type { Config } from '../config/config.js'; import type { ChatRecordingService } from '../services/chatRecordingService.js'; import { createDebugLogger } from '../utils/debugLogger.js'; -import { - evaluateMediaPolicyToolCall, - resolveMediaPolicyModelAccess, -} from '../omni/policy/model-access.js'; +import { evaluateMediaPolicyToolCall } from '../omni/policy/model-access.js'; import { sanitizeToolNameForProvider } from '../utils/tool-name-utils.js'; import { compactToolResultDisplayForHistory } from '../utils/toolResultDisplayCompaction.js'; import { @@ -79,6 +76,7 @@ import { ToolErrorType } from '../tools/tool-error.js'; import { DEFERRED_TOOL_CALL_REFUSAL_PREFIX, DEFERRED_TOOL_CALL_CANCELLATION_PREFIX, + describeBridgedArgumentError, resolveDeferredToolCall, } from '../tools/tool-call.js'; import type { @@ -1093,12 +1091,6 @@ type SchedulerToolCallRequestInfo = ToolCallRequestInfo & { bridgeResolutionError?: { error: Error; type: ToolErrorType; - /** - * Validated bridge target for per-tool parameter-error accounting (the - * wrapper name `tool_call` cannot distinguish targets). Absent when the - * refusal never reached a validated target (e.g. unknown tool). - */ - targetName?: string; }; }; @@ -1108,20 +1100,6 @@ function getModelFacingToolName(request: ToolCallRequestInfo): string { ); } -/** - * Retry-accounting namespace for bridge pre-check refusals. The refusal fires - * before the request is rewritten to the target, but recordRetryableToolError - * prunes same-prefix keys — so keying it on the bare target name would let a - * bridged refusal and a direct validation failure of the SAME target reset - * each other every batch and neither channel would ever reach - * VALIDATION_RETRY_LOOP_THRESHOLD. The suffix keeps the two channels - * independent while the per-batch presence prune still recognizes the entry - * as belonging to the target. - */ -function bridgeRetryToolName(targetName: string): string { - return `${targetName}(via tool_call)`; -} - // NOTE: the `⚠` in this and TRUNCATION_RETRY_LOOP_DIRECTIVE below is part of an // LLM-facing prompt directive (injected into the model prompt, not rendered in // the TUI). The width-1 glyph rationale used elsewhere in this change does not @@ -2803,48 +2781,6 @@ export class CoreToolScheduler { resolveDeferredToolCall(this.toolRegistry, request.args, { // Match prepareTools's depth-gated AgentTool policy. maxSubagentDepth: this.config.getMaxSubagentDepth(), - // A truncated response's arguments are incomplete for transport - // reasons: the pre-check must yield to the truncation guards below. - wasOutputTruncated: request.wasOutputTruncated, - // The owner policy must win over the argument pre-check so a denied - // target keeps its specific EXECUTION_DENIED. - isTargetExecutionAllowed: this.isToolExecutionAllowed, - getDefaultArgumentNames: (targetName) => - Object.keys( - resolveMediaPolicyModelAccess(this.config, targetName) - .defaultArguments, - ), - // The Code Mode and permission-manager gates in _schedule own their - // specific denials; a blocked target skips the argument pre-check - // instead of accruing parameter errors for a call that cannot run. - suppressArgumentPreCheck: async (targetName: string) => { - if ( - this.config.getToolMode?.() === ToolMode.CodeModeOnly && - request.executionOrigin?.kind !== 'fixed_policy' && - !isCodeModeToolCallAllowed( - canonicalToolName(targetName), - request.source ?? 'model', - ) - ) { - return true; - } - if (permissionManager) { - try { - return !(await permissionManager.isToolEnabled(targetName)); - } catch (error) { - // Do not let a policy lookup failure swallow the pre-check. - // On the refusal path _schedule continues ahead of the - // permission gate, so this is the lookup's only record. - debugLogger.warn( - 'Bridge pre-check policy lookup failed for', - targetName, - error, - ); - return false; - } - } - return false; - }, }), ); if ('error' in resolution) { @@ -2853,15 +2789,26 @@ export class CoreToolScheduler { bridgeResolutionError: { error: resolution.error, type: resolution.errorType, - targetName: resolution.targetName, }, }; } - // No post-resolution isToolExecutionAllowed gate here: resolution - // already consulted the same predicate (constructor-fixed) on the same - // target name, so a second copy would be unreachable and free to - // diverge. + // The pre-schedule gates saw the wrapper, so apply the owner's policy to + // the resolved target before execution. + if ( + this.isToolExecutionAllowed && + !this.isToolExecutionAllowed(resolution.tool.name) + ) { + return { + ...request, + bridgeResolutionError: { + error: new Error( + `Tool "${resolution.tool.name}" is not permitted by this agent's tool policy (execution allowlist or disallowedTools blocklist).`, + ), + type: ToolErrorType.EXECUTION_DENIED, + }, + }; + } return { ...request, @@ -2958,15 +2905,12 @@ export class CoreToolScheduler { /** * Removes all validation retry counters for the given tool. Keys are * ":", so a plain `Map.delete(toolName)` would not - * match anything. The bridge-marked channel is cleared too: the two channels - * are one family for presence (see the prune in _schedule), so clearing must - * cover both or a successful execution of the target would leave its stale - * bridge-channel count behind to fire RETRY LOOP DETECTED prematurely. + * match anything. */ private clearRetryCountsForTool(toolName: string): void { - const prefixes = [`${toolName}:`, `${bridgeRetryToolName(toolName)}:`]; + const prefix = `${toolName}:`; for (const key of this.validationRetryCounts.keys()) { - if (prefixes.some((prefix) => key.startsWith(prefix))) { + if (key.startsWith(prefix)) { this.validationRetryCounts.delete(key); } } @@ -3052,40 +2996,8 @@ export class CoreToolScheduler { // whenever any current request matched caused stale counts for // unrelated tools to survive and fire RETRY LOOP DETECTED prematurely // the next time those tools were used. - // - // A target's direct and bridge-marked channels are ONE family for - // presence: a request naming X preserves both `X` and - // `bridgeRetryToolName(X)`. Widening presence only for the bridged - // channel let alternating channels ACROSS batches prune each other — a - // bridged batch kept just the marked key and a direct batch just the - // bare one, so neither counter ever reached - // VALIDATION_RETRY_LOOP_THRESHOLD and a mixed-channel loop rode on - // without the stop directive. clearRetryCountsForTool clears both - // channels, so this widening cannot resurrect the stale bridge count - // that a resolved-and-executed target leaves behind. - // - // A refused bridge request keeps the wrapper name (`tool_call`), so the - // channel-marked name of the validated target must join the presence set - // alongside it — but only for INVALID_TOOL_PARAMS refusals, the one error - // type that accrues below: an EXECUTION_DENIED (policy) refusal records - // nothing, so it must not keep the denied target's stale counters alive - // either. if (this.validationRetryCounts.size > 0) { - const currentToolNames = new Set( - requestsToProcess.flatMap((r) => { - const names = [r.name, bridgeRetryToolName(r.name)]; - if ( - r.bridgeResolutionError?.type === - ToolErrorType.INVALID_TOOL_PARAMS && - r.bridgeResolutionError.targetName !== undefined - ) { - names.push( - bridgeRetryToolName(r.bridgeResolutionError.targetName), - ); - } - return names; - }), - ); + const currentToolNames = new Set(requestsToProcess.map((r) => r.name)); for (const key of [...this.validationRetryCounts.keys()]) { const sep = key.indexOf(':'); const toolName = sep === -1 ? key : key.slice(0, sep); @@ -3165,18 +3077,8 @@ export class CoreToolScheduler { reqInfo.bridgeResolutionError.type === ToolErrorType.INVALID_TOOL_PARAMS ) { - // Key on the validated target, not the `tool_call` wrapper: - // alternating failures against distinct targets must accrue - // per-target instead of pruning each other's counter. The - // channel marker keeps a target's bridged pre-check refusals - // from prefix-pruning (or being pruned by) the SAME target's - // direct validation failures — see bridgeRetryToolName. const count = recordBatchRetryableToolError( - reqInfo.bridgeResolutionError.targetName !== undefined - ? bridgeRetryToolName( - reqInfo.bridgeResolutionError.targetName, - ) - : reqInfo.name, + reqInfo.name, bridgeError.message, ); if (count >= VALIDATION_RETRY_LOOP_THRESHOLD) { @@ -3402,11 +3304,20 @@ export class CoreToolScheduler { ); if (recordPrevalidationCancellation()) continue; if (invocationOrError instanceof Error) { + // A target reached through tool_call reports its own validation + // error; name it so the model does not blame the envelope. + const targetMessage = + reqInfo.modelFacingName !== undefined + ? describeBridgedArgumentError( + reqInfo.name, + invocationOrError.message, + ) + : invocationOrError.message; const displayError = reqInfo.wasOutputTruncated - ? new Error( - `${invocationOrError.message} ${TRUNCATION_PARAM_GUIDANCE}`, - ) - : invocationOrError; + ? new Error(`${targetMessage} ${TRUNCATION_PARAM_GUIDANCE}`) + : reqInfo.modelFacingName !== undefined + ? new Error(targetMessage) + : invocationOrError; // Track validation retry for loop detection. Counts accumulate per // (tool, error message) pair so a different validation mistake on @@ -3418,9 +3329,7 @@ export class CoreToolScheduler { const finalError = count >= VALIDATION_RETRY_LOOP_THRESHOLD - ? new Error( - `${invocationOrError.message}${RETRY_LOOP_STOP_DIRECTIVE}`, - ) + ? new Error(`${targetMessage}${RETRY_LOOP_STOP_DIRECTIVE}`) : displayError; newToolCalls.push({ diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index a817d34432e..674d26c3245 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -314,6 +314,7 @@ export type { CronDeleteTool, CronDeleteParams } from './tools/cron-delete.js'; export { DEFERRED_TOOL_CALL_CANCELLATION_PREFIX, DEFERRED_TOOL_CALL_REFUSAL_PREFIX, + describeBridgedArgumentError, resolveDeferredToolCall, } from './tools/tool-call.js'; export type { diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 8dcb148aab7..5e053a3863b 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -4,7 +4,7 @@ * SPDX-License-Identifier: Apache-2.0 */ -import { describe, expect, it, vi } from 'vitest'; +import { describe, expect, it } from 'vitest'; import { MockTool } from '../test-utils/mock-tool.js'; import { runWithAgentContext } from '../agents/runtime/agent-context.js'; import { runWithTeammateIdentity } from '../agents/team/identity.js'; @@ -12,18 +12,13 @@ import { deferredDeclarationFingerprint, type ToolRegistry, } from './tool-registry.js'; -import type { AnyDeclarativeTool, MediaPolicyToolDescriptor } from './tools.js'; +import type { AnyDeclarativeTool } from './tools.js'; import { DEFERRED_TOOL_CALL_REFUSAL_PREFIX, + describeBridgedArgumentError, resolveDeferredToolCall, ToolCallTool, } from './tool-call.js'; -import { SchemaValidator } from '../utils/schemaValidator.js'; -import { projectMediaPolicyToolDeclaration } from '../omni/policy/model-access.js'; -import { - validateMediaPolicyIoParams, - type MediaPolicyIoParams, -} from '../omni/policy/tools/media-policy-tool.js'; import { ToolErrorType } from './tool-error.js'; import { ToolNames } from './tool-names.js'; import { DEFAULT_MAX_SUBAGENT_DEPTH } from '../config/config.js'; @@ -119,6 +114,19 @@ const NOT_AVAILABLE = refusal( 'not available to this agent', ); +describe('describeBridgedArgumentError', () => { + it('names the target, the bridge and where to read the schema', () => { + expect( + describeBridgedArgumentError( + 'web_fetch', + "params must have required property 'url'.", + ), + ).toBe( + `Deferred tool "web_fetch" (called through ${ToolNames.TOOL_CALL}) rejected the arguments: params must have required property 'url'. Pass arguments matching the schema returned by ${ToolNames.TOOL_SEARCH} for "web_fetch".`, + ); + }); +}); + describe('ToolCallTool', () => { it('is an always-visible bridge with a stable generic schema', () => { const tool = new ToolCallTool(); @@ -422,35 +430,6 @@ describe('ToolCallTool', () => { ); }); - it('denies a policy-blocked target ahead of the hidden-tool gate', async () => { - // Registered NOT hidden: the caller's execution-policy gate must fire - // BEFORE the isDeferredAndHidden check, like the sibling plan-lifecycle - // and leader-only gates above. Otherwise a policy-disabled but visible - // target bridged through tool_call gets the "already visible — call it - // directly" INVALID_TOOL_PARAMS, telling the model to call a tool the - // owner's policy forbids (and on ACP accruing an invalid-parameter - // strike). Mutation check: moving the isTargetExecutionAllowed gate - // below the isDeferredAndHidden check turns this red with the - // "already visible" refusal; dropping targetName from the denial also - // turns it red. - const target = new MockTool({ name: 'web_fetch', shouldDefer: false }); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set()), - { name: target.name, arguments: {} }, - { isTargetExecutionAllowed: async () => false }, - ); - - expect(result).toMatchObject({ - errorType: ToolErrorType.EXECUTION_DENIED, - targetName: 'web_fetch', - error: expect.objectContaining({ - message: expect.stringContaining( - "not permitted by this agent's tool policy", - ), - }), - }); - }); - it.each([ ToolNames.TEAM_DELETE, ToolNames.WORKFLOW, @@ -641,850 +620,4 @@ describe('ToolCallTool', () => { ); } }); - - describe('target-schema pre-validation (#12889)', () => { - // Mirrors the issue's web_fetch: both fields required, so `{}` must not - // be accepted just because the bridge envelope types arguments as a - // bare object. - const makeWebFetchLike = () => - new MockTool({ - name: 'web_fetch', - shouldDefer: true, - params: { - type: 'object', - properties: { - url: { type: 'string' }, - prompt: { type: 'string' }, - }, - required: ['url', 'prompt'], - additionalProperties: false, - }, - }); - - it('resolves a call whose arguments satisfy the target schema', async () => { - const target = makeWebFetchLike(); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { - name: 'web_fetch', - arguments: { url: 'https://example.com', prompt: 'summarize' }, - }, - ); - - expect(result).toMatchObject({ - tool: expect.objectContaining({ name: 'web_fetch' }), - arguments: { url: 'https://example.com', prompt: 'summarize' }, - }); - }); - - it('refuses an empty arguments object that misses required target fields', async () => { - // #12889: the bridge validated only its own envelope, so `{}` passed - // and the target's required-field error surfaced post-unwrap as a bare - // Ajv message the model could not act on. The refusal must name the - // target and the missing field. Mutation check: dropping the - // pre-validation in resolveDeferredToolCall turns this red. - const target = makeWebFetchLike(); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: 'web_fetch', arguments: {} }, - ); - - expect(result).toMatchObject({ - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: 'web_fetch', - }); - expect(result).not.toHaveProperty('tool'); - if ('error' in result) { - expect( - result.error.message.startsWith(DEFERRED_TOOL_CALL_REFUSAL_PREFIX), - ).toBe(true); - expect(result.error.message).toContain('"web_fetch"'); - expect(result.error.message).toContain("'url'"); - } - }); - - it('leaves surplus-key enforcement to the target without changing its schema', async () => { - class LenientTool extends MockTool { - override validateToolParams(): string | null { - return null; - } - } - for (const Tool of [MockTool, LenientTool]) { - const target = new Tool({ - name: 'agent_like', - shouldDefer: true, - params: { - type: 'object', - properties: { prompt: { type: 'string' } }, - required: ['prompt'], - additionalProperties: false, - }, - }); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { - name: target.name, - arguments: { prompt: 'investigate', name: 'helper' }, - }, - ); - expect(result).not.toHaveProperty('error'); - expect(target.schema.parametersJsonSchema).toHaveProperty( - 'additionalProperties', - false, - ); - if ('tool' in result) { - const build = () => result.tool.build(result.arguments); - if (Tool === MockTool) { - expect(build).toThrow('must NOT have additional properties'); - } else { - expect(build).not.toThrow(); - } - } - } - }); - - // Same tolerance one level down: todo_write's item schema declares - // additionalProperties: false while its own validateToolParams only - // type-checks the known keys, so a nested surplus key must not trip the - // pre-check either. - const makeTodoLike = (Tool: typeof MockTool = MockTool) => - new Tool({ - name: 'todo_like', - shouldDefer: true, - params: { - type: 'object', - properties: { - todos: { - type: 'array', - items: { - type: 'object', - properties: { - id: { type: 'string' }, - content: { type: 'string' }, - status: { type: 'string', enum: ['pending', 'completed'] }, - }, - required: ['id', 'content', 'status'], - additionalProperties: false, - }, - }, - }, - required: ['todos'], - additionalProperties: false, - }, - }); - - it('leaves nested surplus-key enforcement to the target without changing its schema', async () => { - // Mutation check: relaxing only the top-level additionalProperties - // turns this red with "must NOT have additional properties". - class LenientTool extends MockTool { - override validateToolParams(): string | null { - return null; - } - } - for (const Tool of [MockTool, LenientTool]) { - const target = makeTodoLike(Tool); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { - name: target.name, - arguments: { - todos: [ - { - id: '1', - content: 'write the test', - status: 'pending', - priority: 'high', - }, - ], - }, - }, - ); - expect(result).not.toHaveProperty('error'); - expect(JSON.stringify(target.schema.parametersJsonSchema)).toContain( - '"additionalProperties":false', - ); - if ('tool' in result) { - const build = () => result.tool.build(result.arguments); - if (Tool === MockTool) { - expect(build).toThrow('must NOT have additional properties'); - } else { - expect(build).not.toThrow(); - } - } - } - }); - - it('still refuses nested schema violations when surplus keys are tolerated', async () => { - const target = makeTodoLike(); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { - name: target.name, - arguments: { - todos: [ - { - id: '1', - content: 'write the test', - status: 'bogus', - priority: 'high', - }, - ], - }, - }, - ); - - expect(result).toMatchObject({ - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: 'todo_like', - }); - if ('error' in result) { - expect( - result.error.message.startsWith(DEFERRED_TOOL_CALL_REFUSAL_PREFIX), - ).toBe(true); - expect(result.error.message).toContain( - 'must be equal to one of the allowed values', - ); - } - }); - - // A deferred and hidden MCP target publishes its server's inputSchema - // unmodified, and per-branch `additionalProperties: false` is the standard - // generated tagged-union idiom: the branches are told apart BY the keyword. - const makeTaggedUnionLike = () => - new MockTool({ - name: 'mcp__srv__union', - shouldDefer: true, - params: { - type: 'object', - oneOf: [ - { - properties: { a: { type: 'string' } }, - required: ['a'], - additionalProperties: false, - }, - { - properties: { a: { type: 'string' }, b: { type: 'string' } }, - required: ['a'], - additionalProperties: false, - }, - ], - }, - }); - - it('resolves a tagged union whose branches are told apart by additionalProperties', async () => { - // R8-1: a rewrite keyed on the property name alone also flipped the - // keyword INSIDE each oneOf branch, so {a, b} matched both branches and - // oneOf (exactly one) failed. The pre-check then refused a call the - // target's own schema and build() both accept, and — because targetName - // is populated — booked a parameter-error strike toward RETRY LOOP - // DETECTED for a tool that works. Mutation check: descending into - // composition keywords turns this red with "must match exactly one - // schema in oneOf". - const target = makeTaggedUnionLike(); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: target.name, arguments: { a: 'x', b: 'y' } }, - ); - - expect(result).not.toHaveProperty('error'); - expect(result).toMatchObject({ arguments: { a: 'x', b: 'y' } }); - // Relaxing happens on a clone: the target's published schema is intact. - expect(JSON.stringify(target.schema.parametersJsonSchema)).toContain( - '"additionalProperties":false', - ); - if ('tool' in result) { - expect(() => result.tool.build(result.arguments)).not.toThrow(); - } - }); - - it('still refuses a tagged-union call that matches no branch', async () => { - // Preserving the branches is not the same as disabling the pre-check: - // `{b}` satisfies neither branch's `required: ['a']`, so it must still be - // refused and attributed to the target. - const target = makeTaggedUnionLike(); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: target.name, arguments: { b: 'y' } }, - ); - - expect(result).toMatchObject({ - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: 'mcp__srv__union', - }); - expect(result).not.toHaveProperty('tool'); - if ('error' in result) { - expect( - result.error.message.startsWith(DEFERRED_TOOL_CALL_REFUSAL_PREFIX), - ).toBe(true); - expect(result.error.message).toContain('"mcp__srv__union"'); - } - }); - - it('leaves annotation data that looks like the keyword alone', async () => { - // Same inversion one class over: `const` holds data the schema compares - // against, so rewriting the keyword inside it makes the pre-check demand - // a value the authored schema rejects. - const target = new MockTool({ - name: 'annotated_target', - shouldDefer: true, - params: { - type: 'object', - properties: { - config: { const: { additionalProperties: false } }, - }, - required: ['config'], - }, - }); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { - name: target.name, - arguments: { config: { additionalProperties: false } }, - }, - ); - - expect(result).not.toHaveProperty('error'); - }); - - it('reads a property named additionalProperties as a name, not the keyword', async () => { - // Keys of a name-to-schema map are data. The authored schema forbids this - // property outright, so the pre-check must keep refusing it rather than - // relax the prohibition away. - const target = new MockTool({ - name: 'named_property_target', - shouldDefer: true, - params: { - type: 'object', - properties: { - prompt: { type: 'string' }, - additionalProperties: false, - }, - required: ['prompt'], - }, - }); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { - name: target.name, - arguments: { prompt: 'investigate', additionalProperties: true }, - }, - ); - - expect(result).toMatchObject({ - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: 'named_property_target', - }); - }); - - it('resolves the same tagged union written with $defs/$ref', async () => { - // The $ref form is what generated schemas actually use. Composition is - // non-lexical through it, so the relaxation must not descend into the - // shared definitions: flipping a branch's additionalProperties there - // makes {a,b} match both and oneOf (exactly one) fails. Mutation check: - // re-adding $defs to the name-keyed walk turns this red with "must match - // exactly one schema in oneOf". - const target = new MockTool({ - name: 'mcp__srv__refunion', - shouldDefer: true, - params: { - type: 'object', - $defs: { - A: { - properties: { a: { type: 'string' } }, - required: ['a'], - additionalProperties: false, - }, - B: { - properties: { a: { type: 'string' }, b: { type: 'string' } }, - required: ['a'], - additionalProperties: false, - }, - }, - oneOf: [{ $ref: '#/$defs/A' }, { $ref: '#/$defs/B' }], - }, - }); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: target.name, arguments: { a: 'x', b: 'y' } }, - ); - - expect(result).not.toHaveProperty('error'); - expect(result).toMatchObject({ arguments: { a: 'x', b: 'y' } }); - if ('tool' in result) { - // The target's own validator (Ajv on the unmodified schema) accepts. - expect(() => result.tool.build(result.arguments)).not.toThrow(); - } - }); - - it('relaxes additionalProperties inside allOf branches, which only widens', async () => { - // allOf branches do not discriminate — both must hold — so a per-branch - // additionalProperties: false is not load-bearing the way oneOf's is. - // args {a,b} fail each strict branch (each forbids the other key) and - // pass both relaxed ones. Mutation check: re-listing allOf as verbatim - // turns this red with a refusal. - class LenientTool extends MockTool { - override validateToolParams(): string | null { - return null; - } - } - const target = new LenientTool({ - name: 'allof_target', - shouldDefer: true, - params: { - type: 'object', - allOf: [ - { - properties: { a: { type: 'string' } }, - required: ['a'], - additionalProperties: false, - }, - { - properties: { b: { type: 'string' } }, - required: ['b'], - additionalProperties: false, - }, - ], - }, - }); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: target.name, arguments: { a: 'x', b: 'y' } }, - ); - - expect(result).not.toHaveProperty('error'); - expect(result).toMatchObject({ arguments: { a: 'x', b: 'y' } }); - }); - - it('never reads a constraint literally named additionalProperties as the keyword', async () => { - // `dependencies` maps a NAME to a constraint; a dependency named - // additionalProperties with a false schema forbids that property, and - // flipping the false to true would invert it into always-pass — the - // pre-check would then resolve a call the target's own validator - // refuses. The constraint must stay false after the walk. - const target = new MockTool({ - name: 'dep_target', - shouldDefer: true, - params: { - type: 'object', - properties: { mode: { type: 'string' } }, - dependencies: { additionalProperties: false }, - }, - }); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: target.name, arguments: { additionalProperties: 'x' } }, - ); - - expect(result).toMatchObject({ - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: 'dep_target', - }); - expect(result).not.toHaveProperty('tool'); - }); - - it('still attributes wrong field types when surplus keys are present', async () => { - const target = makeWebFetchLike(); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { - name: target.name, - arguments: { url: {}, prompt: 'summarize', extra: true }, - }, - ); - expect(result).toMatchObject({ - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: 'web_fetch', - error: expect.objectContaining({ - message: expect.stringContaining('must be string'), - }), - }); - }); - - it('returns the model-sent arguments even when validation coerces a clone', async () => { - // SchemaValidator.validate coerces values in place (numeric strings → - // numbers, etc.). The pre-check must run on a clone: the resolved - // arguments stay exactly what the model sent, and the scheduler - // re-validates them at build time. - const target = new MockTool({ - name: 'counter', - shouldDefer: true, - params: { - type: 'object', - properties: { count: { type: 'integer' } }, - required: ['count'], - }, - }); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: 'counter', arguments: { count: '3' } }, - ); - - expect(result).toMatchObject({ arguments: { count: '3' } }); - }); - - it('leaves optional null placeholders for the target to normalize', async () => { - class NullTolerantTool extends MockTool { - override validateToolParams(params: { - [key: string]: unknown; - }): string | null { - return params['working_dir'] === null - ? null - : super.validateToolParams(params); - } - } - const target = new NullTolerantTool({ - name: 'agent_like', - shouldDefer: true, - params: { - type: 'object', - properties: { - prompt: { type: 'string' }, - working_dir: { type: 'string' }, - }, - required: ['prompt'], - }, - }); - const argumentsWithPlaceholder = { - prompt: 'investigate', - working_dir: null, - }; - - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: target.name, arguments: argumentsWithPlaceholder }, - ); - - expect(result).not.toHaveProperty('error'); - expect(result).toMatchObject({ arguments: argumentsWithPlaceholder }); - }); - - it('does not reserve a target schema id in the shared validator', async () => { - const target = new MockTool({ - name: 'identified_target', - shouldDefer: true, - params: { - $id: 'https://example.com/deferred-tool-precheck', - type: 'object', - properties: { prompt: { type: 'string' } }, - required: ['prompt'], - additionalProperties: false, - }, - }); - - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: target.name, arguments: { prompt: 'investigate' } }, - ); - - expect(result).not.toHaveProperty('error'); - expect(target.validateToolParams({})).toContain("'prompt'"); - }); - - it('re-reads a target schema the target mutates in place after the first call', async () => { - // AgentTool's refresh mutates its own parameterSchema object in place - // (it adds and removes `model`/`name`), and Ajv caches a compiled - // schema by object identity for the life of the process. Handing the - // validator that same object pins every later bridged call to the - // shape the first one happened to compile, ignoring changed constraints. - // A stale validator would ignore the newly advertised model enum and - // accept the unknown grade below. - const target = new MockTool({ - name: 'agent_like', - shouldDefer: true, - params: { - type: 'object', - properties: { prompt: { type: 'string' } }, - required: ['prompt'], - additionalProperties: false, - }, - }); - const registry = makeRegistry([target], new Set([target.name])); - - // The first bridged call compiles the pre-refresh schema. - await resolveDeferredToolCall(registry, { - name: 'agent_like', - arguments: { prompt: 'do the thing' }, - }); - - // The refresh then advertises a new property on that SAME object. - const schema = target.schema.parametersJsonSchema as { - properties: Record; - }; - schema.properties['model'] = { type: 'string', enum: ['fast', 'pro'] }; - - const invalid = await resolveDeferredToolCall(registry, { - name: 'agent_like', - arguments: { prompt: 'do the thing', model: 'unknown-grade' }, - }); - expect(invalid).toMatchObject({ - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: 'agent_like', - }); - - const result = await resolveDeferredToolCall(registry, { - name: 'agent_like', - arguments: { prompt: 'do the thing', model: 'fast' }, - }); - - expect(result).not.toHaveProperty('error'); - expect(result).toMatchObject({ - arguments: { prompt: 'do the thing', model: 'fast' }, - }); - }); - - it('pre-checks only the schema layer, leaving value-level rules to build()', async () => { - // The pre-check exists to name the target and the missing field in - // the refusal (#12889); a target's value-level rules (fs stats, - // content scans, the AgentTool refresh kick) must run exactly once, - // at build() time — running them here would pay their side effects - // twice per bridged call. Mutation check: routing the pre-check - // through target.validateToolParams (schema + value rules) fires the - // spy and turns this red. - const valueRuleSpy = vi.fn( - (_params: { [key: string]: unknown }): string | null => null, - ); - class ValueRuleTool extends MockTool { - protected override validateToolParamValues(params: { - [key: string]: unknown; - }): string | null { - return valueRuleSpy(params); - } - } - const target = new ValueRuleTool({ - name: 'write_file', - shouldDefer: true, - params: { - type: 'object', - properties: { file_path: { type: 'string' } }, - required: ['file_path'], - }, - }); - - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: 'write_file', arguments: { file_path: '/tmp/a.txt' } }, - ); - - expect(result).not.toHaveProperty('error'); - expect(valueRuleSpy).not.toHaveBeenCalled(); - }); - - // The shared native shape of the omni media-policy family: io params - // with `resourceId` as the model-facing `inputPath` alternative, and - // only `outputDir` required natively (0 of the 14 shipped tools require - // inputPath — the call gate resolves resourceId → inputPath before - // build() validates). - const mediaPolicyNativeSchema = { - type: 'object', - properties: { - inputPath: { type: 'string' }, - resourceId: { type: 'string' }, - outputDir: { type: 'string' }, - }, - required: ['outputDir'], - additionalProperties: false, - }; - - // A MockTool carrying the real media-policy split: `schema` is derived - // through the production projector (locked keys stripped from properties - // AND required) instead of a hand-written literal, while - // `validateToolParams` keeps checking the NATIVE schema plus the io - // value rule, exactly like BaseMediaPolicyTool - // (omni/policy/tools/media-policy-tool.ts). - class MockMediaPolicyTool extends MockTool { - constructor(private readonly lockedArguments: Record) { - super({ - name: 'omni_transcribe_audio', - shouldDefer: true, - params: mediaPolicyNativeSchema, - }); - } - - override get mediaPolicyDescriptor(): MediaPolicyToolDescriptor { - return { - kind: 'media_policy', - inputMediaTypes: ['audio'], - outputs: [{ kind: 'media', required: true }], - }; - } - - override get schema() { - return projectMediaPolicyToolDeclaration( - { - getOmniPolicyToolsSettings: () => ({ - [this.name]: { - modelAccess: { - enabled: true, - lockedArguments: this.lockedArguments, - }, - }, - }), - }, - { - name: this.name, - description: this.description, - parametersJsonSchema: mediaPolicyNativeSchema, - }, - ); - } - - override validateToolParams(params: { - [key: string]: unknown; - }): string | null { - return ( - SchemaValidator.validate(mediaPolicyNativeSchema, params) ?? - validateMediaPolicyIoParams(params as unknown as MediaPolicyIoParams) - ); - } - } - - it('resolves a media-policy target whose arguments the policy gate completes', async () => { - // The projection split a media-policy tool creates: `schema` is the - // model-visible declaration (an operator `modelAccess.lockedArguments` - // key stripped from BOTH properties and required), while - // `validateToolParams` keeps checking the NATIVE schema. The model is - // therefore correct to omit `outputDir`, and the modelAccess gate — - // which both frontends run AFTER bridge resolution — merges it back - // in. Pre-checking the raw arguments against the native schema refuses - // a call the next stage accepts, and sending the locked key instead - // makes the gate refuse it: unwinnable both ways. Mutation check: - // dropping the media-policy branch in resolveDeferredToolCall turns - // this red. - const target = new MockMediaPolicyTool({ outputDir: '/locked/out' }); - // The mock really carries the split the defect needs, derived through - // the real projector rather than pinned as a literal: the locked key - // leaves properties and required, the rest of the surface stays. - const projection = target.schema.parametersJsonSchema as { - properties: Record; - required?: string[]; - }; - expect(projection.properties).toHaveProperty('inputPath'); - expect(projection.properties).not.toHaveProperty('outputDir'); - expect(projection.required ?? []).not.toContain('outputDir'); - expect(target.validateToolParams({ inputPath: '/tmp/in.wav' })).toContain( - "'outputDir'", - ); - - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { - name: 'omni_transcribe_audio', - arguments: { inputPath: '/tmp/in.wav' }, - }, - ); - - expect(result).not.toHaveProperty('error'); - expect(result).not.toHaveProperty('errorType'); - expect(result).toMatchObject({ - tool: expect.objectContaining({ name: 'omni_transcribe_audio' }), - arguments: { inputPath: '/tmp/in.wav' }, - }); - }); - - it('resolves a media-policy target called with a resourceId handle and no inputPath', async () => { - // The call shape omni/media-guidance.ts instructs the model to send: - // an opaque session media handle instead of inputPath (the gate - // resolves it to inputPath AFTER bridge resolution), with the locked - // outputDir omitted. The bridge must not apply the native schema or - // the io value rule here — both demand fields only the gate supplies. - // Mutation check: removing the media-policy branch, or pre-checking - // the native schema, refuses this on the locked outputDir. - const target = new MockMediaPolicyTool({ outputDir: '/locked/out' }); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { - name: 'omni_transcribe_audio', - arguments: { resourceId: 'media-1-abcd' }, - }, - ); - - expect(result).not.toHaveProperty('error'); - expect(result).not.toHaveProperty('errorType'); - expect(result).toMatchObject({ - tool: expect.objectContaining({ name: 'omni_transcribe_audio' }), - arguments: { resourceId: 'media-1-abcd' }, - }); - }); - - it('allows gate-supplied defaults without changing the declaration or caller arguments', async () => { - const target = new MockMediaPolicyTool({}); - const schemaBefore = structuredClone(target.schema.parametersJsonSchema); - const registry = makeRegistry([target], new Set([target.name])); - const defaults = vi.fn(() => ['outputDir']); - for (const args of [ - { inputPath: '/tmp/in.wav' }, - { inputPath: '/tmp/in.wav', outputDir: '/caller/out' }, - ]) { - const argsBefore = structuredClone(args); - const result = await resolveDeferredToolCall( - registry, - { name: target.name, arguments: args }, - { getDefaultArgumentNames: defaults }, - ); - expect(result).toMatchObject({ tool: target, arguments: argsBefore }); - expect(args).toEqual(argsBefore); - expect(target.schema.parametersJsonSchema).toEqual(schemaBefore); - } - expect(defaults).toHaveBeenCalledWith(target.name); - const invalid = await resolveDeferredToolCall( - registry, - { name: target.name, arguments: { outputDir: {} } }, - { getDefaultArgumentNames: defaults }, - ); - expect(invalid).toMatchObject({ - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: target.name, - }); - }); - - it('refuses a media-policy target whose arguments miss a model-visible required field', async () => { - // With no lockedArguments the projection is the native schema, so a - // bridged `{}` must still be refused here — naming the target and the - // missing field — instead of surfacing a bare Ajv message from build() - // under the wrapper name. Mutation check: skipping validation for - // media-policy targets turns this red. - const target = new MockMediaPolicyTool({}); - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: 'omni_transcribe_audio', arguments: {} }, - ); - - expect(result).toMatchObject({ - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: 'omni_transcribe_audio', - }); - if ('error' in result) { - expect(result.error.message).toContain('"omni_transcribe_audio"'); - expect(result.error.message).toContain("'outputDir'"); - } - }); - - it('resolves a target whose schema access throws, leaving the throw to build()', async () => { - // A target whose declaration throws under the pre-check must not - // become a new bridge failure mode: the scheduler's build() reports - // the same throw as before. Mutation check: dropping the try/catch - // around the pre-check turns this red. - class ThrowingSchemaTool extends MockTool { - override get schema(): never { - throw new Error('boom from schema access'); - } - } - const target = new ThrowingSchemaTool({ - name: 'throwing_tool', - shouldDefer: true, - params: { type: 'object', properties: {} }, - }); - - const result = await resolveDeferredToolCall( - makeRegistry([target], new Set([target.name])), - { name: 'throwing_tool', arguments: {} }, - ); - - expect(result).not.toHaveProperty('error'); - expect(result).toMatchObject({ - tool: expect.objectContaining({ name: 'throwing_tool' }), - arguments: {}, - }); - }); - }); }); diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 2f3dc841ea2..107812344a2 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -21,7 +21,6 @@ import { deferredDeclarationFingerprint, type ToolRegistry, } from './tool-registry.js'; -import { SchemaValidator } from '../utils/schemaValidator.js'; import { getExcludedToolUnavailableMessage, getLeaderOnlyToolUnavailableMessage, @@ -52,33 +51,6 @@ export type DeferredToolCallResolution = export interface DeferredToolCallOptions { /** Omission keeps AgentTool excluded in subagent contexts. */ maxSubagentDepth?: number; - /** - * Set when the model's response was cut by max_tokens. The bridged - * arguments may be incomplete for transport reasons rather than a schema - * misreading, so the argument pre-check must yield to the caller's - * truncation-aware handling (the scheduler rejects truncated Edit-kind - * calls outright and appends truncation guidance to build-time validation - * failures) instead of reporting a schema mismatch. - */ - wasOutputTruncated?: boolean; - /** - * The caller's per-target execution policy (scheduler execution allowlist - * / permission-manager enablement). Consulted before the argument - * pre-check so a denied target keeps its specific EXECUTION_DENIED refusal - * instead of surfacing a parameter error for a call that could never run. - */ - isTargetExecutionAllowed?: (targetName: string) => boolean | Promise; - /** - * Set when the caller applies its own target policy downstream of - * resolution (the scheduler's permission-manager gate) with a richer - * denial message than the bridge can build. Returning true skips the - * argument pre-check for that target so the caller's own denial — not a - * parameter error for a call that could never run — is what the model - * sees. - */ - suppressArgumentPreCheck?: (targetName: string) => boolean | Promise; - /** Media-policy fields supplied by the caller's downstream modelAccess gate. */ - getDefaultArgumentNames?: (targetName: string) => readonly string[]; } export const DEFERRED_TOOL_CALL_REFUSAL_PREFIX = '[tool_call bridge refused] '; @@ -90,97 +62,17 @@ function bridgeRefusal(message: string): Error { } /** - * Schema keywords whose subtree the relaxation below must leave byte-identical. - * `oneOf`/`not` discriminate: a per-branch `additionalProperties: false` tells - * branches apart, so relaxing it there inverts the schema's meaning instead of - * widening acceptance (`if` selects a branch by the same mechanism). Annotation - * keywords hold data the schema compares against, not a subschema. `$defs` and - * `definitions` are reached only through `$ref`: they are shared definitions, - * and relaxing inside one silently rewrites every branch that references it — - * each use site is already covered directly by the walk above. - * (`allOf`/`anyOf`/`then`/`else` are deliberately absent: relaxing under them - * only widens acceptance, so the walk descends.) + * Names the target in a validation error its own `build()` raised after + * tool_call unwrapped it. Unlabelled, "params must have required property + * 'url'" reads as a fault in tool_call's `{name, arguments}` envelope + * (#12889). Only relabels a rejection the target already made, so it can never + * refuse a call the target would accept. */ -const VERBATIM_SCHEMA_KEYS: ReadonlySet = new Set([ - 'oneOf', - 'not', - 'if', - 'const', - 'default', - 'enum', - 'example', - 'examples', - '$defs', - 'definitions', -]); - -/** - * Schema keywords whose value maps an arbitrary NAME to a subschema or - * constraint. The names are data, so a property literally named - * `additionalProperties` keeps its own schema rather than being read as the - * keyword: these are walked by value only. - */ -const NAME_TO_SCHEMA_KEYS: ReadonlySet = new Set([ - 'dependencies', - 'dependentSchemas', - 'patternProperties', - 'properties', -]); - -/** Relaxes `additionalProperties: false` in an already-cloned schema tree. */ -function relaxAdditionalPropertiesInPlace(node: unknown): void { - if (Array.isArray(node)) { - for (const item of node) { - relaxAdditionalPropertiesInPlace(item); - } - return; - } - if (!node || typeof node !== 'object') { - return; - } - const schema = node as Record; - for (const [key, value] of Object.entries(schema)) { - if (VERBATIM_SCHEMA_KEYS.has(key)) { - continue; - } - if (NAME_TO_SCHEMA_KEYS.has(key)) { - if (value && typeof value === 'object' && !Array.isArray(value)) { - const byName = value as Record; - for (const subschema of Object.values(byName)) { - relaxAdditionalPropertiesInPlace(subschema); - } - } - continue; - } - if (key === 'additionalProperties' && value === false) { - schema[key] = true; - continue; - } - relaxAdditionalPropertiesInPlace(value); - } -} - -/** - * Deep-clones a target schema with `additionalProperties: false` relaxed at - * every position where relaxing only widens acceptance. Some targets - * deliberately tolerate surplus keys (for example, Agent's `name` outside team - * mode, or todo_write's item-level extra fields), so that decision is left to - * their own build(): a nested `additionalProperties: false` (todo_write's items - * schema) must not refuse bridged calls the target's own validator accepts. - * - * The walk is structural rather than keyed on the property name alone, because - * a name-keyed rewrite also reaches the positions listed in - * `VERBATIM_SCHEMA_KEYS` and `NAME_TO_SCHEMA_KEYS`, where it makes the - * pre-check STRICTER than the schema the target publishes: an input matching - * exactly one `oneOf` branch then matches two and `oneOf` fails, and a `const` - * branch compares against rewritten data. - */ -function relaxAdditionalProperties(schema: unknown): Record { - // The JSON round-trip deep-clones, so the walk can relax in place and never - // touches the target's own schema object (which it may mutate and reuse). - const clone = JSON.parse(JSON.stringify(schema)) as Record; - relaxAdditionalPropertiesInPlace(clone); - return clone; +export function describeBridgedArgumentError( + targetName: string, + message: string, +): string { + return `Deferred tool "${targetName}" (called through ${ToolNames.TOOL_CALL}) rejected the arguments: ${message.replace(/\.$/, '')}. Pass arguments matching the schema returned by ${ToolNames.TOOL_SEARCH} for "${targetName}".`; } export async function resolveDeferredToolCall( @@ -302,21 +194,6 @@ export async function resolveDeferredToolCall( errorType: ToolErrorType.EXECUTION_DENIED, }; } - // The caller's execution policy precedes the hidden-tool gate and the - // argument pre-check: a denied target keeps its specific denial rather - // than a parameter error for a call that could never run. - if ( - options?.isTargetExecutionAllowed !== undefined && - !(await options.isTargetExecutionAllowed(target.name)) - ) { - return { - error: bridgeRefusal( - `Tool "${target.name}" is not permitted by this agent's tool policy (execution allowlist or disallowedTools blocklist).`, - ), - errorType: ToolErrorType.EXECUTION_DENIED, - targetName: target.name, - }; - } if (!registry.isDeferredAndHidden(target.name)) { return { @@ -388,93 +265,6 @@ export async function resolveDeferredToolCall( }; } - // The bridge envelope deliberately types `arguments` as a bare object (the - // declaration must stay byte-stable across catalog changes), so `{}` is - // envelope-valid even when the target requires fields. Pre-validate against - // the target's model-visible schema so the refusal names the target and the - // missing field, instead of surfacing a bare Ajv message after the call has - // been unwrapped (#12889). Validate clones of both sides: SchemaValidator - // coerces argument values in place and the scheduler re-validates them at - // build time; and Ajv caches a compiled schema by object identity, so a - // target that mutates its own schema object in place (AgentTool's refresh - // adds and removes `model`/`name`) would otherwise stay pinned to whatever - // shape it had on the first bridged call. Compile the per-call copy in an - // isolated validator so it sees the current shape without reserving the - // target's `$id` in the process-shared registry. - // - // Only the schema layer runs here — never the target's full - // validateToolParams: its value-level rules (fs stats, content scans, the - // AgentTool refresh kick) run unchanged at build() time, so running them - // here would pay their side effects twice per bridged call. The - // model-visible `schema` getter is also what makes this safe for omni - // media-policy targets: their declaration is a projection with operator - // `lockedArguments` stripped from `required`, while their - // `validateToolParams` deliberately checks the NATIVE schema plus io value - // rules that assume the modelAccess gate (which both frontends run AFTER - // bridge resolution) has merged those arguments back in — running it here - // would refuse calls the very next stage accepts. - // Defaults remain model-visible and overridable; omit their names only - // from this clone's required list because the same gate supplies them. - let paramsError: string | null = null; - // A truncated response yields to the caller's truncation handling: the - // arguments are incomplete for transport reasons, not a schema misreading. - // A target the caller's own downstream policy gate will deny (the - // scheduler's permission-manager gate owns the richer denial message) - // must surface that denial, not a parameter error for a call that could - // never run. - const preCheckSuppressed = - options?.suppressArgumentPreCheck !== undefined && - (await options.suppressArgumentPreCheck(target.name)); - if (!options?.wasOutputTruncated && !preCheckSuppressed) { - try { - const argsClone = structuredClone(invocation.params.arguments); - // Surplus-key tolerance is the target's own call, so relax the keyword - // wherever relaxing only widens acceptance — but never inside a - // composition branch or annotation data, where the rewrite inverts the - // schema's meaning and the pre-check ends up stricter than the schema the - // target publishes. See relaxAdditionalProperties. - const schemaClone = relaxAdditionalProperties( - target.schema.parametersJsonSchema, - ); - if ( - target.mediaPolicyDescriptor?.kind === 'media_policy' && - Array.isArray(schemaClone['required']) - ) { - const defaults = new Set( - options?.getDefaultArgumentNames?.(target.name), - ); - schemaClone['required'] = schemaClone['required'].filter( - (name) => !defaults.has(name), - ); - } - const required = new Set( - Array.isArray(schemaClone['required']) ? schemaClone['required'] : [], - ); - for (const [name, value] of Object.entries(argsClone)) { - if (value === null && !required.has(name)) { - delete argsClone[name]; - } - } - const compiled = SchemaValidator.compileIsolated(schemaClone); - if ('validate' in compiled) { - paramsError = compiled.validate(argsClone); - } - } catch { - // A target whose validation throws under this pre-check must not become - // a new bridge failure mode: the scheduler's build() reports the same - // throw as before. - } - } - if (paramsError) { - return { - error: bridgeRefusal( - `Deferred tool "${target.name}" rejected the arguments: ${paramsError}. Pass arguments matching the schema returned by tool_search for "${target.name}".`, - ), - errorType: ToolErrorType.INVALID_TOOL_PARAMS, - targetName: target.name, - }; - } - return { tool: target, arguments: structuredClone(invocation.params.arguments), From 403da277605bfbb8be476a672743ae55b61d90e2 Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Fri, 2 Oct 2026 00:26:36 +0900 Subject: [PATCH 19/27] fix(core): let a Responses model fill tool_call arguments; declare an unreachable target #12889: through an OpenAI Responses-compatible provider, a model kept calling tool_call with {"name":"web_fetch","arguments":{}}. It repeated that after a clear "missing url" error until loop protection stopped the turn. A clearer error alone (the previous head) did not change it. Root cause, on the Responses wire only. normalizeResponsesSchemaNode patches every object without `properties` to `properties: {}` for Azure. For the bridge's nested `arguments: {type: "object"}` that turns "any object" into "an empty object", and a backend that constrains decoding to the schema can then produce nothing but {}. The Chat Completions and Anthropic converters have no such patch, which matches where #12889 reproduces. A nested node patched this way now also gets an explicit `additionalProperties: true` unless it states one. The root keeps the bare patch, since a zero-arg tool's top level really is empty. Fallback, for any provider or model that still sends nothing. When a hidden deferred target reached through tool_call received no arguments at all and its own validation rejected that, the primary session declares it directly (revealDeferredTool + setTools, rolled back if the refresh fails). The error then says to call it by name with its real schema. Any non-empty arguments never trigger it, so an ordinary argument mistake does not grow the declaration list. Subagents are excluded, and the cost is one prompt-prefix rebuild on a path that was already failing. Wired in both the scheduler and the ACP session. Not run here: an E2E against the reporter's provider, which is the acceptance criterion on #12889. --- .../src/acp-integration/session/Session.ts | 27 ++++-- .../core/src/core/coreToolScheduler.test.ts | 43 +++++++++ packages/core/src/core/coreToolScheduler.ts | 19 +++- .../responses-converter.test.ts | 58 +++++++++++-- .../responses-converter.ts | 16 +++- packages/core/src/index.ts | 2 + packages/core/src/tools/tool-call.test.ts | 87 ++++++++++++++++++- packages/core/src/tools/tool-call.ts | 44 ++++++++++ 8 files changed, 278 insertions(+), 18 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.ts b/packages/cli/src/acp-integration/session/Session.ts index 76fcf1fba20..536603a2e42 100644 --- a/packages/cli/src/acp-integration/session/Session.ts +++ b/packages/cli/src/acp-integration/session/Session.ts @@ -112,7 +112,9 @@ import { ToolErrorType, DEFERRED_TOOL_CALL_REFUSAL_PREFIX, DEFERRED_TOOL_CALL_CANCELLATION_PREFIX, + declareTargetAfterEmptyBridgedCall, describeBridgedArgumentError, + describeDirectDeclaration, resolveDeferredToolCall, CreateSubSessionTool, fireNotificationHook, @@ -15949,13 +15951,24 @@ export class Session implements SessionContext { } catch (e) { const caught = e instanceof Error ? e : new Error(String(e)); // Same labelling as the scheduler: a target reached through - // tool_call names itself when its own build() rejects the arguments. - const error = - bridgedThroughToolCall && !toolBuildSucceeded - ? new Error( - describeBridgedArgumentError(toolName, caught.message), - ) - : caught; + // tool_call names itself when its own build() rejects the arguments, + // and is declared directly when the bridge delivered none (#12889). + const bridgedBuildFailure = + bridgedThroughToolCall && !toolBuildSucceeded; + const declaredDirectly = + bridgedBuildFailure && + (await declareTargetAfterEmptyBridgedCall( + this.config.getToolRegistry(), + this.config.getLlmClient?.(), + toolName, + args, + )); + const error = bridgedBuildFailure + ? new Error( + describeBridgedArgumentError(toolName, caught.message) + + (declaredDirectly ? describeDirectDeclaration(toolName) : ''), + ) + : caught; const hooksEnabledForError = !this.config.getDisableAllHooks?.(); const messageBusForError = this.config.getMessageBus?.(); const executionTimeoutException = diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index 7f9b8711022..e1350dff75f 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -1195,6 +1195,7 @@ describe('CoreToolScheduler', () => { hasMatchingAskRule?: (ctx: unknown) => boolean; }; deferredHiddenNames?: ReadonlySet; + revealDeferredTool?: (name: string) => void; includeToolSearch?: boolean; isToolExecutionAllowed?: (name: string) => boolean; }) { @@ -1232,6 +1233,7 @@ describe('CoreToolScheduler', () => { getAllToolNames: () => [...options.toolsByName.keys()], isDeferredAndHidden: (name: string) => options.deferredHiddenNames?.has(name) ?? false, + revealDeferredTool: options.revealDeferredTool, }), { getApprovalMode: () => options.approvalMode ?? ApprovalMode.YOLO, @@ -1696,6 +1698,47 @@ describe('CoreToolScheduler', () => { expect(message).toContain(ToolNames.TOOL_SEARCH); }); + it('declares the target directly when the bridge delivered no arguments (#12889)', async () => { + const revealDeferredTool = vi.fn(); + const setTools = vi.fn(async () => {}); + const { completed, deferred } = await runBridgeCall( + 'bridge-empty-args', + { params: URL_REQUIRED_PARAMS }, + { revealDeferredTool, getLlmClient: () => ({ setTools }) }, + ); + + expectStatus(completed, 'error'); + expect(revealDeferredTool).toHaveBeenCalledWith(deferred.name); + expect(setTools).toHaveBeenCalledOnce(); + const message = completed.response.error?.message ?? ''; + expect(message).toContain("must have required property 'url'"); + expect(message).toContain(`"${deferred.name}" is now declared directly`); + }); + + it('keeps the target hidden when the bridged arguments are present but wrong', async () => { + const revealDeferredTool = vi.fn(); + const setTools = vi.fn(async () => {}); + const harness = bridgeWithDeferred( + { params: URL_REQUIRED_PARAMS }, + { revealDeferredTool, getLlmClient: () => ({ setTools }) }, + ); + + await scheduleBridgeCall( + harness.scheduler, + 'bridge-wrong-args', + harness.deferred.name, + { url: 42 }, + ); + + const completed = firstBatch(harness.onAllToolCallsComplete)[0]; + expectStatus(completed, 'error'); + expect(revealDeferredTool).not.toHaveBeenCalled(); + expect(setTools).not.toHaveBeenCalled(); + expect(completed.response.error?.message ?? '').not.toContain( + 'declared directly', + ); + }); + it("leaves a direct call's validation error unlabelled", async () => { const direct = new MockTool({ name: 'needs_url', diff --git a/packages/core/src/core/coreToolScheduler.ts b/packages/core/src/core/coreToolScheduler.ts index 527cd831f3a..f7253ca3ee9 100644 --- a/packages/core/src/core/coreToolScheduler.ts +++ b/packages/core/src/core/coreToolScheduler.ts @@ -76,7 +76,9 @@ import { ToolErrorType } from '../tools/tool-error.js'; import { DEFERRED_TOOL_CALL_REFUSAL_PREFIX, DEFERRED_TOOL_CALL_CANCELLATION_PREFIX, + declareTargetAfterEmptyBridgedCall, describeBridgedArgumentError, + describeDirectDeclaration, resolveDeferredToolCall, } from '../tools/tool-call.js'; import type { @@ -3305,13 +3307,26 @@ export class CoreToolScheduler { if (recordPrevalidationCancellation()) continue; if (invocationOrError instanceof Error) { // A target reached through tool_call reports its own validation - // error; name it so the model does not blame the envelope. + // error; name it so the model does not blame the envelope. If the + // bridge delivered no arguments at all, also declare the target + // directly so the model can fill its real schema (#12889). + const declaredDirectly = + reqInfo.modelFacingName !== undefined && + (await declareTargetAfterEmptyBridgedCall( + this.toolRegistry, + this.config.getLlmClient?.(), + reqInfo.name, + reqInfo.args, + )); const targetMessage = reqInfo.modelFacingName !== undefined ? describeBridgedArgumentError( reqInfo.name, invocationOrError.message, - ) + ) + + (declaredDirectly + ? describeDirectDeclaration(reqInfo.name) + : '') : invocationOrError.message; const displayError = reqInfo.wasOutputTruncated ? new Error(`${targetMessage} ${TRUNCATION_PARAM_GUIDANCE}`) diff --git a/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.test.ts b/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.test.ts index fd9009454dc..3b2f639cede 100644 --- a/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.test.ts +++ b/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.test.ts @@ -28,6 +28,7 @@ import type { ResponsesApiReasoningItem, ResponsesSSEEvent, } from './types.js'; +import { ToolCallTool } from '../../tools/tool-call.js'; import { getGenAiUsageProvenance } from '../../telemetry/gen-ai-usage.js'; import { getThoughtSummary } from '../../utils/thoughtUtils.js'; import { content, fnCall, userText } from '../../test-utils/model-fixtures.js'; @@ -1069,10 +1070,13 @@ describe('normalizeResponsesParameters', () => { expect(normalizeResponsesParameters(schema)).toEqual({ type: 'object', properties: { - nested: { type: 'object', properties: {} }, - list: { type: 'array', items: { type: 'object', properties: {} } }, + nested: { type: 'object', properties: {}, additionalProperties: true }, + list: { + type: 'array', + items: { type: 'object', properties: {}, additionalProperties: true }, + }, }, - anyOf: [{ type: 'object', properties: {} }], + anyOf: [{ type: 'object', properties: {}, additionalProperties: true }], }); }); @@ -1086,8 +1090,8 @@ describe('normalizeResponsesParameters', () => { expect(normalizeResponsesParameters(schema)).toEqual({ type: 'object', properties: {}, - oneOf: [{ type: 'object', properties: {} }], - allOf: [{ type: 'object', properties: {} }], + oneOf: [{ type: 'object', properties: {}, additionalProperties: true }], + allOf: [{ type: 'object', properties: {}, additionalProperties: true }], }); }); @@ -1113,13 +1117,55 @@ describe('normalizeResponsesParameters', () => { expect(result).not.toBe(schema); expect(result).toEqual({ type: 'object', - properties: { nested: { type: 'object', properties: {} } }, + properties: { + nested: { type: 'object', properties: {}, additionalProperties: true }, + }, }); }); it('passes through undefined', () => { expect(normalizeResponsesParameters(undefined)).toBeUndefined(); }); + + it('keeps a nested open object open, and an explicit closed one closed', () => { + // A nested `{type:'object'}` means "any object"; `properties: {}` alone + // reads as "an empty object" to a backend that constrains decoding to + // the schema (#12889). + expect( + normalizeResponsesParameters({ + type: 'object', + properties: { + open: { type: 'object' }, + closed: { type: 'object', additionalProperties: false }, + typed: { type: 'object', additionalProperties: { type: 'string' } }, + }, + }), + ).toEqual({ + type: 'object', + properties: { + open: { type: 'object', properties: {}, additionalProperties: true }, + closed: { type: 'object', properties: {}, additionalProperties: false }, + typed: { + type: 'object', + properties: {}, + additionalProperties: { type: 'string' }, + }, + }, + }); + }); + + it("keeps tool_call's arguments open on the Responses wire (#12889)", () => { + const bridge = new ToolCallTool().schema; + const normalized = normalizeResponsesParameters( + bridge.parametersJsonSchema as Record, + ) as { properties: Record> }; + + expect(normalized.properties['arguments']).toMatchObject({ + type: 'object', + properties: {}, + additionalProperties: true, + }); + }); }); describe('assistant phase replay', () => { diff --git a/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.ts b/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.ts index b86804c040d..5d8cf176019 100644 --- a/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.ts +++ b/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.ts @@ -874,16 +874,25 @@ export function convertGeminiToolsToResponsesTools( * helper patches such schemas (including nested object schemas) to include * `properties: {}` so Azure accepts them. Well-formed schemas pass through * unchanged. + * + * A nested object without `properties` means "any object" (JSON Schema's + * `additionalProperties` defaults to true). With only `properties: {}` added, + * a backend that constrains decoding to the schema reads it as "an empty + * object" and can produce nothing else — which is how the deferred-tool + * bridge's `tool_call.arguments` reached a Responses model as `{}` on every + * attempt (#12889). So a nested node patched here also gets an explicit + * `additionalProperties: true` unless it already states one. The root keeps + * the bare patch: a zero-arg tool's top level really is empty. */ export function normalizeResponsesParameters( schema: Record | undefined, ): Record | undefined { if (schema === undefined) return undefined; if (schema === null || typeof schema !== 'object') return schema; - return normalizeResponsesSchemaNode(schema) as Record; + return normalizeResponsesSchemaNode(schema, true) as Record; } -function normalizeResponsesSchemaNode(node: unknown): unknown { +function normalizeResponsesSchemaNode(node: unknown, isRoot = false): unknown { if (Array.isArray(node)) { return node.map((item) => normalizeResponsesSchemaNode(item)); } @@ -913,6 +922,9 @@ function normalizeResponsesSchemaNode(node: unknown): unknown { if (out['type'] === 'object' && out['properties'] === undefined) { out['properties'] = {}; + if (!isRoot && out['additionalProperties'] === undefined) { + out['additionalProperties'] = true; + } } return out; diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 674d26c3245..7237b28d766 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -314,7 +314,9 @@ export type { CronDeleteTool, CronDeleteParams } from './tools/cron-delete.js'; export { DEFERRED_TOOL_CALL_CANCELLATION_PREFIX, DEFERRED_TOOL_CALL_REFUSAL_PREFIX, + declareTargetAfterEmptyBridgedCall, describeBridgedArgumentError, + describeDirectDeclaration, resolveDeferredToolCall, } from './tools/tool-call.js'; export type { diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 5e053a3863b..cb3a2b6e5e9 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -4,7 +4,7 @@ * SPDX-License-Identifier: Apache-2.0 */ -import { describe, expect, it } from 'vitest'; +import { describe, expect, it, vi } from 'vitest'; import { MockTool } from '../test-utils/mock-tool.js'; import { runWithAgentContext } from '../agents/runtime/agent-context.js'; import { runWithTeammateIdentity } from '../agents/team/identity.js'; @@ -15,7 +15,9 @@ import { import type { AnyDeclarativeTool } from './tools.js'; import { DEFERRED_TOOL_CALL_REFUSAL_PREFIX, + declareTargetAfterEmptyBridgedCall, describeBridgedArgumentError, + describeDirectDeclaration, resolveDeferredToolCall, ToolCallTool, } from './tool-call.js'; @@ -127,6 +129,89 @@ describe('describeBridgedArgumentError', () => { }); }); +describe('declareTargetAfterEmptyBridgedCall (#12889)', () => { + function revealRegistry(hidden = true) { + return { + isDeferredAndHidden: vi.fn(() => hidden), + revealDeferredTool: vi.fn(), + unrevealDeferredTool: vi.fn(), + }; + } + const declare = ( + registry: ReturnType, + client: { setTools(): Promise } | null, + args: Record | undefined, + ) => + declareTargetAfterEmptyBridgedCall( + registry as unknown as ToolRegistry, + client, + 'web_fetch', + args, + ); + + it('declares a hidden target that the bridge delivered no arguments to', async () => { + const registry = revealRegistry(); + const setTools = vi.fn(async () => {}); + + await expect(declare(registry, { setTools }, {})).resolves.toBe(true); + await expect(declare(registry, { setTools }, undefined)).resolves.toBe( + true, + ); + expect(registry.revealDeferredTool).toHaveBeenCalledWith('web_fetch'); + expect(setTools).toHaveBeenCalledTimes(2); + }); + + it('leaves the declaration list alone for an ordinary argument mistake', async () => { + const registry = revealRegistry(); + const setTools = vi.fn(async () => {}); + + await expect(declare(registry, { setTools }, { url: 42 })).resolves.toBe( + false, + ); + expect(registry.revealDeferredTool).not.toHaveBeenCalled(); + expect(setTools).not.toHaveBeenCalled(); + }); + + it('does nothing for a visible target, without a client, or in a subagent', async () => { + const setTools = vi.fn(async () => {}); + const visible = revealRegistry(false); + await expect(declare(visible, { setTools }, {})).resolves.toBe(false); + + const noClient = revealRegistry(); + await expect(declare(noClient, null, {})).resolves.toBe(false); + + const subagent = revealRegistry(); + await expect( + runWithAgentContext('worker', () => declare(subagent, { setTools }, {})), + ).resolves.toBe(false); + + for (const registry of [visible, noClient, subagent]) { + expect(registry.revealDeferredTool).not.toHaveBeenCalled(); + } + expect(setTools).not.toHaveBeenCalled(); + }); + + it('rolls the reveal back when the declaration refresh fails', async () => { + const registry = revealRegistry(); + const setTools = vi.fn(async () => { + throw new Error('refresh failed'); + }); + + await expect(declare(registry, { setTools }, {})).resolves.toBe(false); + expect(registry.revealDeferredTool).toHaveBeenCalledWith('web_fetch'); + expect(registry.unrevealDeferredTool).toHaveBeenCalledWith('web_fetch'); + }); + + it('tells the model to call the declared tool by name', () => { + expect(describeDirectDeclaration('web_fetch')).toContain( + 'call "web_fetch" by name', + ); + expect(describeDirectDeclaration('web_fetch')).toContain( + `not through ${ToolNames.TOOL_CALL}`, + ); + }); +}); + describe('ToolCallTool', () => { it('is an always-visible bridge with a stable generic schema', () => { const tool = new ToolCallTool(); diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 107812344a2..ab71ad41519 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -75,6 +75,50 @@ export function describeBridgedArgumentError( return `Deferred tool "${targetName}" (called through ${ToolNames.TOOL_CALL}) rejected the arguments: ${message.replace(/\.$/, '')}. Pass arguments matching the schema returned by ${ToolNames.TOOL_SEARCH} for "${targetName}".`; } +/** + * Recovery for a hidden target that a model could not reach through the + * bridge (#12889). `tool_call` declares `arguments` as an open object, so a + * model that cannot fill it — a constrained decoder, or a model that ignores + * the target schema it reviewed — keeps sending `{}` and stops on the retry + * guard. When a hidden deferred target reached through tool_call received no + * arguments at all and its own validation rejected that, declare it directly + * so the next request carries its real parameter schema. + * + * Primary session only: a subagent's declarations are not the LlmClient's. + * Narrow on purpose — an ordinary argument mistake (any non-empty arguments) + * never widens the declaration list, because a revealed tool stays declared + * for the rest of the session. The cost is one prompt-prefix rebuild, paid + * only on this failure path. Returns whether the target is now declared. + */ +export async function declareTargetAfterEmptyBridgedCall( + registry: ToolRegistry, + client: { setTools(): Promise } | null | undefined, + targetName: string, + args: Record | undefined, +): Promise { + if (args && Object.keys(args).length > 0) return false; + if (!client || isSubagentLikeExecutionContext()) return false; + if ( + typeof registry.revealDeferredTool !== 'function' || + !registry.isDeferredAndHidden(targetName) + ) { + return false; + } + registry.revealDeferredTool(targetName); + try { + await client.setTools(); + } catch { + registry.unrevealDeferredTool?.(targetName); + return false; + } + return true; +} + +/** Appended to the bridged validation error once the target is declared. */ +export function describeDirectDeclaration(targetName: string): string { + return ` "${targetName}" is now declared directly with its full parameter schema: call "${targetName}" by name with its required arguments, not through ${ToolNames.TOOL_CALL}.`; +} + export async function resolveDeferredToolCall( registry: ToolRegistry, envelope: Record, From d7465d12ab38147635422ba653db5f515e27cb2b Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Fri, 2 Oct 2026 00:53:03 +0900 Subject: [PATCH 20/27] test(core): use non-empty arguments that still fail validation {url: 42} passed: the schema validator coerces 42 to "42", so the call succeeded and the test never reached the no-reveal branch. Send a non-empty object without the required url instead. --- packages/core/src/core/coreToolScheduler.test.ts | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index e1350dff75f..6bab4b0d54d 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -1727,7 +1727,9 @@ describe('CoreToolScheduler', () => { harness.scheduler, 'bridge-wrong-args', harness.deferred.name, - { url: 42 }, + // Non-empty but still missing `url`. A wrong type is no good here: the + // schema validator coerces `42` to `"42"` and the call succeeds. + { title: 'news' }, ); const completed = firstBatch(harness.onAllToolCallsComplete)[0]; From 8b1b24791fbc7d2a4c3e4ae6a437376415dff760 Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Fri, 2 Oct 2026 00:24:58 +0800 Subject: [PATCH 21/27] fix(core): keep deferred tool recovery in the primary session --- packages/core/src/tools/tool-call.test.ts | 20 +++++++++++++++++++- packages/core/src/tools/tool-call.ts | 4 +++- 2 files changed, 22 insertions(+), 2 deletions(-) diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index cb3a2b6e5e9..412c139ad12 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -6,7 +6,12 @@ import { describe, expect, it, vi } from 'vitest'; import { MockTool } from '../test-utils/mock-tool.js'; -import { runWithAgentContext } from '../agents/runtime/agent-context.js'; +import { + getCurrentAgentId, + runWithAgentChat, + runWithAgentContext, +} from '../agents/runtime/agent-context.js'; +import type { LlmChat } from '../core/llm-chat.js'; import { runWithTeammateIdentity } from '../agents/team/identity.js'; import { deferredDeclarationFingerprint, @@ -191,6 +196,19 @@ describe('declareTargetAfterEmptyBridgedCall (#12889)', () => { expect(setTools).not.toHaveBeenCalled(); }); + it('keeps primary declarations unchanged in an agent chat without an agent ID', async () => { + const registry = revealRegistry(); + const setTools = vi.fn(async () => {}); + + await runWithAgentChat({} as LlmChat, async () => { + expect(getCurrentAgentId()).toBeNull(); + await expect(declare(registry, { setTools }, {})).resolves.toBe(false); + }); + + expect(registry.revealDeferredTool).not.toHaveBeenCalled(); + expect(setTools).not.toHaveBeenCalled(); + }); + it('rolls the reveal back when the declaration refresh fails', async () => { const registry = revealRegistry(); const setTools = vi.fn(async () => { diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index ab71ad41519..92a5ef05325 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -11,6 +11,7 @@ import type { } from './tools.js'; import { BaseDeclarativeTool, BaseToolInvocation, Kind } from './tools.js'; import { ToolErrorType } from './tool-error.js'; +import { getCurrentAgentChat } from '../agents/runtime/agent-context.js'; import { canonicalToolName, resolveRegisteredToolName, @@ -97,7 +98,8 @@ export async function declareTargetAfterEmptyBridgedCall( args: Record | undefined, ): Promise { if (args && Object.keys(args).length > 0) return false; - if (!client || isSubagentLikeExecutionContext()) return false; + if (!client || getCurrentAgentChat() || isSubagentLikeExecutionContext()) + return false; if ( typeof registry.revealDeferredTool !== 'function' || !registry.isDeferredAndHidden(targetName) From 9af1deeb25e269c94785d00b26f220858e30c563 Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Fri, 2 Oct 2026 09:53:10 +0800 Subject: [PATCH 22/27] fix(core): Preserve eager tool demotion during bridge recovery --- packages/core/src/tools/tool-call.test.ts | 12 ++++++++++++ packages/core/src/tools/tool-call.ts | 1 + 2 files changed, 13 insertions(+) diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 412c139ad12..0557f039d8c 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -177,6 +177,18 @@ describe('declareTargetAfterEmptyBridgedCall (#12889)', () => { expect(setTools).not.toHaveBeenCalled(); }); + it('keeps a target demoted by tools.eager behind the bridge', async () => { + const registry = { + ...revealRegistry(), + isPermissionDeferred: vi.fn(() => true), + }; + const setTools = vi.fn(async () => {}); + + await expect(declare(registry, { setTools }, {})).resolves.toBe(false); + expect(registry.revealDeferredTool).not.toHaveBeenCalled(); + expect(setTools).not.toHaveBeenCalled(); + }); + it('does nothing for a visible target, without a client, or in a subagent', async () => { const setTools = vi.fn(async () => {}); const visible = revealRegistry(false); diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 92a5ef05325..83e3e7f6776 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -102,6 +102,7 @@ export async function declareTargetAfterEmptyBridgedCall( return false; if ( typeof registry.revealDeferredTool !== 'function' || + registry.isPermissionDeferred?.(targetName) === true || !registry.isDeferredAndHidden(targetName) ) { return false; From 20930cd0c249375f3933311c7b6643cf53980e29 Mon Sep 17 00:00:00 2001 From: yiliang114 Date: Fri, 2 Oct 2026 16:22:55 +0800 Subject: [PATCH 23/27] fix(core): guard bridged empty-arg recovery against cancellation and truncation Two round-13 Criticals on the #12889 recovery path in _schedule: R13-1: the `await declareTargetAfterEmptyBridgedCall(...)` was the only await in the loop with no following recordPrevalidationCancellation() re-check. An abort during client.setTools() booked an INVALID_TOOL_PARAMS error instead of a cancellation, and the retry strike survived the batch (clearRetryCountsForTool runs only on a successful validation), so a later genuine retry of the same tool reached VALIDATION_RETRY_LOOP_THRESHOLD one call early. The re-check uses the sibling form: the helper pushes the cancelled response itself. R13-2: a MAX_TOKENS turn marks every pending call truncated (turn.ts), and a truncated `arguments` buffer falls back to {} via safeJsonParse (streamingToolCallParser.ts, responses-converter.ts), so empty args do not prove the model cannot fill the schema. A reveal stays declared for the rest of the session, so truncation must not buy one; this mirrors the sibling truncation gate above. Target naming is kept, and because describeDirectDeclaration is conditional on the reveal, gating it also drops the "call it by name, not through tool_call" directive that contradicted TRUNCATION_PARAM_GUIDANCE. Both are pinned by new cases in coreToolScheduler.test.ts that fail on the unfixed source. Co-authored-by: Qwen-Coder Patrol-Run: qwen-pr-closeout/jmuqn1gku0t --- .../core/src/core/coreToolScheduler.test.ts | 87 +++++++++++++++++++ packages/core/src/core/coreToolScheduler.ts | 20 +++++ 2 files changed, 107 insertions(+) diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index b45f86c2d33..70497e2d597 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -1741,6 +1741,93 @@ describe('CoreToolScheduler', () => { ); }); + it('keeps the target hidden when the empty bridged arguments came from a truncated turn', async () => { + // R13-2: turn.ts marks EVERY pending call of a MAX_TOKENS turn truncated, + // and a truncated `arguments` buffer falls back to `{}` + // (streamingToolCallParser.ts, responses-converter.ts), so empty args do + // not prove the model cannot fill the schema. A reveal stays declared for + // the rest of the session, so a transient truncation must not buy one. + const revealDeferredTool = vi.fn(); + const setTools = vi.fn(async () => {}); + const harness = bridgeWithDeferred( + { params: URL_REQUIRED_PARAMS }, + { revealDeferredTool, getLlmClient: () => ({ setTools }) }, + ); + + await harness.scheduler.schedule( + { + ...toolRequest( + 'bridge-truncated-empty-args', + ToolNames.TOOL_CALL, + { name: harness.deferred.name, arguments: {} }, + 'prompt-bridge-truncated', + ), + wasOutputTruncated: true, + }, + new AbortController().signal, + ); + + const completed = firstBatch(harness.onAllToolCallsComplete)[0]; + expectStatus(completed, 'error'); + expect(revealDeferredTool).not.toHaveBeenCalled(); + expect(setTools).not.toHaveBeenCalled(); + const message = completed.response.error?.message ?? ''; + // The target naming and the truncation diagnosis both survive; only the + // "now declared directly ... not through tool_call" directive — which + // would contradict "retry the tool call" — goes with the reveal. + expect(message).toContain(`"${harness.deferred.name}"`); + expect(message).toContain("must have required property 'url'"); + expect(message).not.toContain('declared directly'); + expect(message).toContain('truncated due to max_tokens'); + }); + + it('books a cancellation when the turn is aborted during the declaration refresh', async () => { + // R13-1: the declaration refresh awaits client.setTools() -> warmAll(), + // and was this loop's only await with no sibling + // recordPrevalidationCancellation() re-check. An abort inside it must book + // a cancellation, not an INVALID_TOOL_PARAMS error. + const abortController = new AbortController(); + const revealDeferredTool = vi.fn(); + const setTools = vi.fn(async () => { + abortController.abort(); + }); + const { completed, deferred, scheduler, onAllToolCallsComplete } = + await runBridgeCall( + 'bridge-abort-during-refresh', + { params: URL_REQUIRED_PARAMS }, + { revealDeferredTool, getLlmClient: () => ({ setTools }) }, + abortController.signal, + ); + + expectStatus(completed, 'cancelled'); + expect(completed.response.errorType).not.toBe( + ToolErrorType.INVALID_TOOL_PARAMS, + ); + // The refresh itself is not rolled back: it completed before the abort was + // observed, and declareTargetAfterEmptyBridgedCall only unreveals when + // setTools throws. + expect(setTools).toHaveBeenCalledOnce(); + expect(revealDeferredTool).toHaveBeenCalledWith(deferred.name); + + // The cancelled turn booked no retry strike. THRESHOLD - 1 further + // identical failures therefore stay under the loop guard; without the + // re-check the cancelled call is the first strike and the last of these + // injects RETRY LOOP DETECTED one call early. + for (const callId of ['bridge-abort-strike-1', 'bridge-abort-strike-2']) { + const later = await completeBridgeCall( + scheduler, + onAllToolCallsComplete, + callId, + deferred.name, + 'prompt-bridge-abort-strike', + ); + expectStatus(later, 'error'); + expect(later.response.error?.message ?? '').not.toContain( + 'RETRY LOOP DETECTED', + ); + } + }); + it("leaves a direct call's validation error unlabelled", async () => { const direct = new MockTool({ name: 'needs_url', diff --git a/packages/core/src/core/coreToolScheduler.ts b/packages/core/src/core/coreToolScheduler.ts index 9f2469a24e9..f6f00ba4005 100644 --- a/packages/core/src/core/coreToolScheduler.ts +++ b/packages/core/src/core/coreToolScheduler.ts @@ -3321,14 +3321,34 @@ export class CoreToolScheduler { // error; name it so the model does not blame the envelope. If the // bridge delivered no arguments at all, also declare the target // directly so the model can fill its real schema (#12889). + // + // A truncated turn is not that state: turn.ts marks *every* pending + // call of a MAX_TOKENS turn, and a truncated `arguments` buffer + // falls back to `{}` (streamingToolCallParser.ts, + // responses-converter.ts), so empty args here do not prove the + // model cannot fill the schema. A reveal stays declared for the + // rest of the session, so truncation must not buy one; the + // TRUNCATION_PARAM_GUIDANCE below already drives the retry. The + // target naming stays — it is accurate either way, and dropping + // `describeDirectDeclaration` with the reveal removes the "call it + // by name, not through tool_call" directive that would contradict + // that guidance. const declaredDirectly = reqInfo.modelFacingName !== undefined && + !reqInfo.wasOutputTruncated && (await declareTargetAfterEmptyBridgedCall( this.toolRegistry, this.config.getLlmClient?.(), reqInfo.name, reqInfo.args, )); + // The declaration refresh above is this loop's only await with no + // sibling re-check: it reaches client.setTools() -> warmAll(). An + // abort inside it must book a cancellation, not an + // INVALID_TOOL_PARAMS error whose retry strike survives the batch + // (clearRetryCountsForTool runs only on a successful validation) + // and trips VALIDATION_RETRY_LOOP_THRESHOLD one call early later. + if (recordPrevalidationCancellation()) continue; const targetMessage = reqInfo.modelFacingName !== undefined ? describeBridgedArgumentError( From b4f63ddb874ff03837d2032ac9a010d7c1a1dcdd Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Fri, 2 Oct 2026 19:08:57 +0800 Subject: [PATCH 24/27] fix(acp): Keep truncated bridged calls hidden --- .../acp-integration/session/Session.test.ts | 78 +++++++++++++++++++ .../src/acp-integration/session/Session.ts | 25 ++++-- 2 files changed, 98 insertions(+), 5 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.test.ts b/packages/cli/src/acp-integration/session/Session.test.ts index 6dd6f6f85e3..a4580a6a2de 100644 --- a/packages/cli/src/acp-integration/session/Session.test.ts +++ b/packages/cli/src/acp-integration/session/Session.test.ts @@ -18458,6 +18458,84 @@ describe('Session', () => { }, ); + it.each(['STOP', 'MAX_TOKENS'])( + 'keeps ACP bridge recovery scoped to non-truncated output (%s)', + async (finishReason) => { + const [{ ToolCallTool }, { ToolSearchTool }, { WebFetchTool }] = + await Promise.all([ + import('@qwen-code/qwen-code-core/tools/tool-call.js'), + import('@qwen-code/qwen-code-core/tools/tool-search.js'), + import('@qwen-code/qwen-code-core/tools/web-fetch.js'), + ]); + const target = new WebFetchTool(mockConfig); + const tools = [ + new ToolCallTool(), + new ToolSearchTool(mockConfig), + target, + ]; + const findTool = (name: string) => + tools.find((tool) => tool.name === name); + mockToolRegistry.getTool.mockImplementation(findTool); + mockToolRegistry.ensureTool.mockImplementation(async (name: string) => + findTool(name), + ); + let hidden = true; + mockToolRegistry.isDeferredAndHidden.mockImplementation( + (name: string) => name === target.name && hidden, + ); + mockToolRegistry.revealDeferredTool.mockImplementation(() => { + hidden = false; + }); + mockChat.sendMessageStream = vi + .fn() + .mockResolvedValueOnce( + createStreamWithChunks([ + { + type: core.StreamEventType.CHUNK, + value: { + functionCalls: [ + { + id: 'empty-bridge', + name: core.ToolNames.TOOL_CALL, + args: { name: target.name, arguments: {} }, + }, + ], + }, + }, + { + type: core.StreamEventType.CHUNK, + value: { candidates: [{ finishReason }] }, + }, + ]), + ) + .mockImplementation(async () => createEmptyStream()); + mockLlmClient.setTools.mockClear(); + + await session.prompt({ + sessionId: 'test-session-id', + prompt: [{ type: 'text', text: 'fetch the news' }], + }); + + const shouldReveal = finishReason === 'STOP'; + expect(hidden).toBe(!shouldReveal); + expect(mockToolRegistry.revealDeferredTool).toHaveBeenCalledTimes( + shouldReveal ? 1 : 0, + ); + expect(mockLlmClient.setTools).toHaveBeenCalledTimes( + shouldReveal ? 1 : 0, + ); + const output = JSON.stringify( + mockChatRecordingService.recordToolResult.mock.calls, + ); + expect(output).toContain('Deferred tool'); + expect(output).toContain(target.name); + expect(output).toContain('url'); + expect(output.includes('is now declared directly')).toBe( + shouldReveal, + ); + }, + ); + it('routes tool_call through a hidden deferred tool in ACP', async () => { mockConfig.getApprovalMode = vi.fn().mockReturnValue(ApprovalMode.YOLO); const execute = vi.fn().mockResolvedValue({ diff --git a/packages/cli/src/acp-integration/session/Session.ts b/packages/cli/src/acp-integration/session/Session.ts index 78b593a5d57..f9e611b996e 100644 --- a/packages/cli/src/acp-integration/session/Session.ts +++ b/packages/cli/src/acp-integration/session/Session.ts @@ -19,11 +19,12 @@ import { randomUUID } from 'node:crypto'; import { realpathSync, statSync } from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; -import type { - Content, - FunctionCall, - GenerateContentResponseUsageMetadata, - Part, +import { + FinishReason, + type Content, + type FunctionCall, + type GenerateContentResponseUsageMetadata, + type Part, } from '@google/genai'; import { type AgentRunContext, @@ -2153,6 +2154,7 @@ function isCallerCausedModelRefusal(error: Error): boolean { */ export class Session implements SessionContext { private readonly mcpAppCalls = new Map(); + private readonly truncatedToolCalls = new WeakSet(); private pendingPrompt: AbortController | null = null; private activeGoalProposalTurn?: AgentResponseCapture['goalProposalTurn']; /** @@ -8927,7 +8929,10 @@ export class Session implements SessionContext { llmClient.discardManagedAutoMemoryRecallDelivery(memoryDelivery); return { responseStream: null, stopReason: 'end_turn' }; } + const truncatedToolCalls = this.truncatedToolCalls; const responseStream = (async function* () { + const functionCalls: FunctionCall[] = []; + let wasOutputTruncated = false; let committed = false; let receivedChunk = false; let memoryDeliveryStateInvalidated = false; @@ -8942,6 +8947,10 @@ export class Session implements SessionContext { for await (const event of sourceStream) { if (event.type === StreamEventType.CHUNK) { receivedChunk = true; + functionCalls.push(...(event.value.functionCalls ?? [])); + wasOutputTruncated ||= + event.value.candidates?.[0]?.finishReason === + FinishReason.MAX_TOKENS; } else if (event.type === StreamEventType.COMPRESSED) { llmClient.resetManagedAutoMemoryAfterCompression(); memoryDeliveryStateInvalidated = true; @@ -8950,6 +8959,8 @@ export class Session implements SessionContext { event.type === StreamEventType.MODEL_FALLBACK ) { receivedChunk = false; + functionCalls.length = 0; + wasOutputTruncated = false; } yield event; } @@ -8957,6 +8968,9 @@ export class Session implements SessionContext { commitMemoryDelivery(); } } finally { + if (wasOutputTruncated) { + for (const call of functionCalls) truncatedToolCalls.add(call); + } if (!committed && receivedChunk && abortSignal.aborted) { commitMemoryDelivery(); } @@ -15964,6 +15978,7 @@ export class Session implements SessionContext { bridgedThroughToolCall && !toolBuildSucceeded; const declaredDirectly = bridgedBuildFailure && + !this.truncatedToolCalls.has(fc) && (await declareTargetAfterEmptyBridgedCall( this.config.getToolRegistry(), this.config.getLlmClient?.(), From a69f8d46c96e90cdec768d426b026c374fe77453 Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Fri, 2 Oct 2026 20:33:53 +0900 Subject: [PATCH 25/27] revert(core): drop the empty-argument auto-reveal fallback from the bridge Removes the fallback that declared a hidden deferred target after an empty-arguments tool_call, together with the four follow-up fixes it needed (subagent exclusion, tools.eager demotion, cancellation and truncation guards, ACP truncation). Each fix closed one review Critical and exposed the next (R12-1, R13-1, R13-2). Kept: the target-named validation error (scheduler and ACP) and the Responses converter fix that stops nested tool_call arguments from being narrowed to an empty object, which is the likely root cause of #12889. The removed code is preserved at archive/12901-empty-arg-auto-reveal. --- .../acp-integration/session/Session.test.ts | 78 ----------- .../src/acp-integration/session/Session.ts | 52 ++----- .../core/src/core/coreToolScheduler.test.ts | 132 ------------------ packages/core/src/core/coreToolScheduler.ts | 39 +----- packages/core/src/index.ts | 2 - packages/core/src/tools/tool-call.test.ts | 119 +--------------- packages/core/src/tools/tool-call.ts | 47 ------- 7 files changed, 16 insertions(+), 453 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.test.ts b/packages/cli/src/acp-integration/session/Session.test.ts index a4580a6a2de..6dd6f6f85e3 100644 --- a/packages/cli/src/acp-integration/session/Session.test.ts +++ b/packages/cli/src/acp-integration/session/Session.test.ts @@ -18458,84 +18458,6 @@ describe('Session', () => { }, ); - it.each(['STOP', 'MAX_TOKENS'])( - 'keeps ACP bridge recovery scoped to non-truncated output (%s)', - async (finishReason) => { - const [{ ToolCallTool }, { ToolSearchTool }, { WebFetchTool }] = - await Promise.all([ - import('@qwen-code/qwen-code-core/tools/tool-call.js'), - import('@qwen-code/qwen-code-core/tools/tool-search.js'), - import('@qwen-code/qwen-code-core/tools/web-fetch.js'), - ]); - const target = new WebFetchTool(mockConfig); - const tools = [ - new ToolCallTool(), - new ToolSearchTool(mockConfig), - target, - ]; - const findTool = (name: string) => - tools.find((tool) => tool.name === name); - mockToolRegistry.getTool.mockImplementation(findTool); - mockToolRegistry.ensureTool.mockImplementation(async (name: string) => - findTool(name), - ); - let hidden = true; - mockToolRegistry.isDeferredAndHidden.mockImplementation( - (name: string) => name === target.name && hidden, - ); - mockToolRegistry.revealDeferredTool.mockImplementation(() => { - hidden = false; - }); - mockChat.sendMessageStream = vi - .fn() - .mockResolvedValueOnce( - createStreamWithChunks([ - { - type: core.StreamEventType.CHUNK, - value: { - functionCalls: [ - { - id: 'empty-bridge', - name: core.ToolNames.TOOL_CALL, - args: { name: target.name, arguments: {} }, - }, - ], - }, - }, - { - type: core.StreamEventType.CHUNK, - value: { candidates: [{ finishReason }] }, - }, - ]), - ) - .mockImplementation(async () => createEmptyStream()); - mockLlmClient.setTools.mockClear(); - - await session.prompt({ - sessionId: 'test-session-id', - prompt: [{ type: 'text', text: 'fetch the news' }], - }); - - const shouldReveal = finishReason === 'STOP'; - expect(hidden).toBe(!shouldReveal); - expect(mockToolRegistry.revealDeferredTool).toHaveBeenCalledTimes( - shouldReveal ? 1 : 0, - ); - expect(mockLlmClient.setTools).toHaveBeenCalledTimes( - shouldReveal ? 1 : 0, - ); - const output = JSON.stringify( - mockChatRecordingService.recordToolResult.mock.calls, - ); - expect(output).toContain('Deferred tool'); - expect(output).toContain(target.name); - expect(output).toContain('url'); - expect(output.includes('is now declared directly')).toBe( - shouldReveal, - ); - }, - ); - it('routes tool_call through a hidden deferred tool in ACP', async () => { mockConfig.getApprovalMode = vi.fn().mockReturnValue(ApprovalMode.YOLO); const execute = vi.fn().mockResolvedValue({ diff --git a/packages/cli/src/acp-integration/session/Session.ts b/packages/cli/src/acp-integration/session/Session.ts index f9e611b996e..70ffd329937 100644 --- a/packages/cli/src/acp-integration/session/Session.ts +++ b/packages/cli/src/acp-integration/session/Session.ts @@ -19,12 +19,11 @@ import { randomUUID } from 'node:crypto'; import { realpathSync, statSync } from 'node:fs'; import * as os from 'node:os'; import * as path from 'node:path'; -import { - FinishReason, - type Content, - type FunctionCall, - type GenerateContentResponseUsageMetadata, - type Part, +import type { + Content, + FunctionCall, + GenerateContentResponseUsageMetadata, + Part, } from '@google/genai'; import { type AgentRunContext, @@ -113,9 +112,7 @@ import { ToolErrorType, DEFERRED_TOOL_CALL_REFUSAL_PREFIX, DEFERRED_TOOL_CALL_CANCELLATION_PREFIX, - declareTargetAfterEmptyBridgedCall, describeBridgedArgumentError, - describeDirectDeclaration, resolveDeferredToolCall, CreateSubSessionTool, fireNotificationHook, @@ -2154,7 +2151,6 @@ function isCallerCausedModelRefusal(error: Error): boolean { */ export class Session implements SessionContext { private readonly mcpAppCalls = new Map(); - private readonly truncatedToolCalls = new WeakSet(); private pendingPrompt: AbortController | null = null; private activeGoalProposalTurn?: AgentResponseCapture['goalProposalTurn']; /** @@ -8929,10 +8925,7 @@ export class Session implements SessionContext { llmClient.discardManagedAutoMemoryRecallDelivery(memoryDelivery); return { responseStream: null, stopReason: 'end_turn' }; } - const truncatedToolCalls = this.truncatedToolCalls; const responseStream = (async function* () { - const functionCalls: FunctionCall[] = []; - let wasOutputTruncated = false; let committed = false; let receivedChunk = false; let memoryDeliveryStateInvalidated = false; @@ -8947,10 +8940,6 @@ export class Session implements SessionContext { for await (const event of sourceStream) { if (event.type === StreamEventType.CHUNK) { receivedChunk = true; - functionCalls.push(...(event.value.functionCalls ?? [])); - wasOutputTruncated ||= - event.value.candidates?.[0]?.finishReason === - FinishReason.MAX_TOKENS; } else if (event.type === StreamEventType.COMPRESSED) { llmClient.resetManagedAutoMemoryAfterCompression(); memoryDeliveryStateInvalidated = true; @@ -8959,8 +8948,6 @@ export class Session implements SessionContext { event.type === StreamEventType.MODEL_FALLBACK ) { receivedChunk = false; - functionCalls.length = 0; - wasOutputTruncated = false; } yield event; } @@ -8968,9 +8955,6 @@ export class Session implements SessionContext { commitMemoryDelivery(); } } finally { - if (wasOutputTruncated) { - for (const call of functionCalls) truncatedToolCalls.add(call); - } if (!committed && receivedChunk && abortSignal.aborted) { commitMemoryDelivery(); } @@ -15972,25 +15956,13 @@ export class Session implements SessionContext { } catch (e) { const caught = e instanceof Error ? e : new Error(String(e)); // Same labelling as the scheduler: a target reached through - // tool_call names itself when its own build() rejects the arguments, - // and is declared directly when the bridge delivered none (#12889). - const bridgedBuildFailure = - bridgedThroughToolCall && !toolBuildSucceeded; - const declaredDirectly = - bridgedBuildFailure && - !this.truncatedToolCalls.has(fc) && - (await declareTargetAfterEmptyBridgedCall( - this.config.getToolRegistry(), - this.config.getLlmClient?.(), - toolName, - args, - )); - const error = bridgedBuildFailure - ? new Error( - describeBridgedArgumentError(toolName, caught.message) + - (declaredDirectly ? describeDirectDeclaration(toolName) : ''), - ) - : caught; + // tool_call names itself when its own build() rejects the arguments. + const error = + bridgedThroughToolCall && !toolBuildSucceeded + ? new Error( + describeBridgedArgumentError(toolName, caught.message), + ) + : caught; const hooksEnabledForError = !this.config.getDisableAllHooks?.(); const messageBusForError = this.config.getMessageBus?.(); const executionTimeoutException = diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index 70497e2d597..6c746c6ead7 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -1195,7 +1195,6 @@ describe('CoreToolScheduler', () => { hasMatchingAskRule?: (ctx: unknown) => boolean; }; deferredHiddenNames?: ReadonlySet; - revealDeferredTool?: (name: string) => void; includeToolSearch?: boolean; isToolExecutionAllowed?: (name: string) => boolean; }) { @@ -1233,7 +1232,6 @@ describe('CoreToolScheduler', () => { getAllToolNames: () => [...options.toolsByName.keys()], isDeferredAndHidden: (name: string) => options.deferredHiddenNames?.has(name) ?? false, - revealDeferredTool: options.revealDeferredTool, }), { getApprovalMode: () => options.approvalMode ?? ApprovalMode.YOLO, @@ -1698,136 +1696,6 @@ describe('CoreToolScheduler', () => { expect(message).toContain(ToolNames.TOOL_SEARCH); }); - it('declares the target directly when the bridge delivered no arguments (#12889)', async () => { - const revealDeferredTool = vi.fn(); - const setTools = vi.fn(async () => {}); - const { completed, deferred } = await runBridgeCall( - 'bridge-empty-args', - { params: URL_REQUIRED_PARAMS }, - { revealDeferredTool, getLlmClient: () => ({ setTools }) }, - ); - - expectStatus(completed, 'error'); - expect(revealDeferredTool).toHaveBeenCalledWith(deferred.name); - expect(setTools).toHaveBeenCalledOnce(); - const message = completed.response.error?.message ?? ''; - expect(message).toContain("must have required property 'url'"); - expect(message).toContain(`"${deferred.name}" is now declared directly`); - }); - - it('keeps the target hidden when the bridged arguments are present but wrong', async () => { - const revealDeferredTool = vi.fn(); - const setTools = vi.fn(async () => {}); - const harness = bridgeWithDeferred( - { params: URL_REQUIRED_PARAMS }, - { revealDeferredTool, getLlmClient: () => ({ setTools }) }, - ); - - await scheduleBridgeCall( - harness.scheduler, - 'bridge-wrong-args', - harness.deferred.name, - // Non-empty but still missing `url`. A wrong type is no good here: the - // schema validator coerces `42` to `"42"` and the call succeeds. - { title: 'news' }, - ); - - const completed = firstBatch(harness.onAllToolCallsComplete)[0]; - expectStatus(completed, 'error'); - expect(revealDeferredTool).not.toHaveBeenCalled(); - expect(setTools).not.toHaveBeenCalled(); - expect(completed.response.error?.message ?? '').not.toContain( - 'declared directly', - ); - }); - - it('keeps the target hidden when the empty bridged arguments came from a truncated turn', async () => { - // R13-2: turn.ts marks EVERY pending call of a MAX_TOKENS turn truncated, - // and a truncated `arguments` buffer falls back to `{}` - // (streamingToolCallParser.ts, responses-converter.ts), so empty args do - // not prove the model cannot fill the schema. A reveal stays declared for - // the rest of the session, so a transient truncation must not buy one. - const revealDeferredTool = vi.fn(); - const setTools = vi.fn(async () => {}); - const harness = bridgeWithDeferred( - { params: URL_REQUIRED_PARAMS }, - { revealDeferredTool, getLlmClient: () => ({ setTools }) }, - ); - - await harness.scheduler.schedule( - { - ...toolRequest( - 'bridge-truncated-empty-args', - ToolNames.TOOL_CALL, - { name: harness.deferred.name, arguments: {} }, - 'prompt-bridge-truncated', - ), - wasOutputTruncated: true, - }, - new AbortController().signal, - ); - - const completed = firstBatch(harness.onAllToolCallsComplete)[0]; - expectStatus(completed, 'error'); - expect(revealDeferredTool).not.toHaveBeenCalled(); - expect(setTools).not.toHaveBeenCalled(); - const message = completed.response.error?.message ?? ''; - // The target naming and the truncation diagnosis both survive; only the - // "now declared directly ... not through tool_call" directive — which - // would contradict "retry the tool call" — goes with the reveal. - expect(message).toContain(`"${harness.deferred.name}"`); - expect(message).toContain("must have required property 'url'"); - expect(message).not.toContain('declared directly'); - expect(message).toContain('truncated due to max_tokens'); - }); - - it('books a cancellation when the turn is aborted during the declaration refresh', async () => { - // R13-1: the declaration refresh awaits client.setTools() -> warmAll(), - // and was this loop's only await with no sibling - // recordPrevalidationCancellation() re-check. An abort inside it must book - // a cancellation, not an INVALID_TOOL_PARAMS error. - const abortController = new AbortController(); - const revealDeferredTool = vi.fn(); - const setTools = vi.fn(async () => { - abortController.abort(); - }); - const { completed, deferred, scheduler, onAllToolCallsComplete } = - await runBridgeCall( - 'bridge-abort-during-refresh', - { params: URL_REQUIRED_PARAMS }, - { revealDeferredTool, getLlmClient: () => ({ setTools }) }, - abortController.signal, - ); - - expectStatus(completed, 'cancelled'); - expect(completed.response.errorType).not.toBe( - ToolErrorType.INVALID_TOOL_PARAMS, - ); - // The refresh itself is not rolled back: it completed before the abort was - // observed, and declareTargetAfterEmptyBridgedCall only unreveals when - // setTools throws. - expect(setTools).toHaveBeenCalledOnce(); - expect(revealDeferredTool).toHaveBeenCalledWith(deferred.name); - - // The cancelled turn booked no retry strike. THRESHOLD - 1 further - // identical failures therefore stay under the loop guard; without the - // re-check the cancelled call is the first strike and the last of these - // injects RETRY LOOP DETECTED one call early. - for (const callId of ['bridge-abort-strike-1', 'bridge-abort-strike-2']) { - const later = await completeBridgeCall( - scheduler, - onAllToolCallsComplete, - callId, - deferred.name, - 'prompt-bridge-abort-strike', - ); - expectStatus(later, 'error'); - expect(later.response.error?.message ?? '').not.toContain( - 'RETRY LOOP DETECTED', - ); - } - }); - it("leaves a direct call's validation error unlabelled", async () => { const direct = new MockTool({ name: 'needs_url', diff --git a/packages/core/src/core/coreToolScheduler.ts b/packages/core/src/core/coreToolScheduler.ts index f6f00ba4005..3069f951c51 100644 --- a/packages/core/src/core/coreToolScheduler.ts +++ b/packages/core/src/core/coreToolScheduler.ts @@ -76,9 +76,7 @@ import { ToolErrorType } from '../tools/tool-error.js'; import { DEFERRED_TOOL_CALL_REFUSAL_PREFIX, DEFERRED_TOOL_CALL_CANCELLATION_PREFIX, - declareTargetAfterEmptyBridgedCall, describeBridgedArgumentError, - describeDirectDeclaration, resolveDeferredToolCall, } from '../tools/tool-call.js'; import type { @@ -3318,46 +3316,13 @@ export class CoreToolScheduler { if (recordPrevalidationCancellation()) continue; if (invocationOrError instanceof Error) { // A target reached through tool_call reports its own validation - // error; name it so the model does not blame the envelope. If the - // bridge delivered no arguments at all, also declare the target - // directly so the model can fill its real schema (#12889). - // - // A truncated turn is not that state: turn.ts marks *every* pending - // call of a MAX_TOKENS turn, and a truncated `arguments` buffer - // falls back to `{}` (streamingToolCallParser.ts, - // responses-converter.ts), so empty args here do not prove the - // model cannot fill the schema. A reveal stays declared for the - // rest of the session, so truncation must not buy one; the - // TRUNCATION_PARAM_GUIDANCE below already drives the retry. The - // target naming stays — it is accurate either way, and dropping - // `describeDirectDeclaration` with the reveal removes the "call it - // by name, not through tool_call" directive that would contradict - // that guidance. - const declaredDirectly = - reqInfo.modelFacingName !== undefined && - !reqInfo.wasOutputTruncated && - (await declareTargetAfterEmptyBridgedCall( - this.toolRegistry, - this.config.getLlmClient?.(), - reqInfo.name, - reqInfo.args, - )); - // The declaration refresh above is this loop's only await with no - // sibling re-check: it reaches client.setTools() -> warmAll(). An - // abort inside it must book a cancellation, not an - // INVALID_TOOL_PARAMS error whose retry strike survives the batch - // (clearRetryCountsForTool runs only on a successful validation) - // and trips VALIDATION_RETRY_LOOP_THRESHOLD one call early later. - if (recordPrevalidationCancellation()) continue; + // error; name it so the model does not blame the envelope. const targetMessage = reqInfo.modelFacingName !== undefined ? describeBridgedArgumentError( reqInfo.name, invocationOrError.message, - ) + - (declaredDirectly - ? describeDirectDeclaration(reqInfo.name) - : '') + ) : invocationOrError.message; const displayError = reqInfo.wasOutputTruncated ? new Error(`${targetMessage} ${TRUNCATION_PARAM_GUIDANCE}`) diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 7237b28d766..674d26c3245 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -314,9 +314,7 @@ export type { CronDeleteTool, CronDeleteParams } from './tools/cron-delete.js'; export { DEFERRED_TOOL_CALL_CANCELLATION_PREFIX, DEFERRED_TOOL_CALL_REFUSAL_PREFIX, - declareTargetAfterEmptyBridgedCall, describeBridgedArgumentError, - describeDirectDeclaration, resolveDeferredToolCall, } from './tools/tool-call.js'; export type { diff --git a/packages/core/src/tools/tool-call.test.ts b/packages/core/src/tools/tool-call.test.ts index 0557f039d8c..5e053a3863b 100644 --- a/packages/core/src/tools/tool-call.test.ts +++ b/packages/core/src/tools/tool-call.test.ts @@ -4,14 +4,9 @@ * SPDX-License-Identifier: Apache-2.0 */ -import { describe, expect, it, vi } from 'vitest'; +import { describe, expect, it } from 'vitest'; import { MockTool } from '../test-utils/mock-tool.js'; -import { - getCurrentAgentId, - runWithAgentChat, - runWithAgentContext, -} from '../agents/runtime/agent-context.js'; -import type { LlmChat } from '../core/llm-chat.js'; +import { runWithAgentContext } from '../agents/runtime/agent-context.js'; import { runWithTeammateIdentity } from '../agents/team/identity.js'; import { deferredDeclarationFingerprint, @@ -20,9 +15,7 @@ import { import type { AnyDeclarativeTool } from './tools.js'; import { DEFERRED_TOOL_CALL_REFUSAL_PREFIX, - declareTargetAfterEmptyBridgedCall, describeBridgedArgumentError, - describeDirectDeclaration, resolveDeferredToolCall, ToolCallTool, } from './tool-call.js'; @@ -134,114 +127,6 @@ describe('describeBridgedArgumentError', () => { }); }); -describe('declareTargetAfterEmptyBridgedCall (#12889)', () => { - function revealRegistry(hidden = true) { - return { - isDeferredAndHidden: vi.fn(() => hidden), - revealDeferredTool: vi.fn(), - unrevealDeferredTool: vi.fn(), - }; - } - const declare = ( - registry: ReturnType, - client: { setTools(): Promise } | null, - args: Record | undefined, - ) => - declareTargetAfterEmptyBridgedCall( - registry as unknown as ToolRegistry, - client, - 'web_fetch', - args, - ); - - it('declares a hidden target that the bridge delivered no arguments to', async () => { - const registry = revealRegistry(); - const setTools = vi.fn(async () => {}); - - await expect(declare(registry, { setTools }, {})).resolves.toBe(true); - await expect(declare(registry, { setTools }, undefined)).resolves.toBe( - true, - ); - expect(registry.revealDeferredTool).toHaveBeenCalledWith('web_fetch'); - expect(setTools).toHaveBeenCalledTimes(2); - }); - - it('leaves the declaration list alone for an ordinary argument mistake', async () => { - const registry = revealRegistry(); - const setTools = vi.fn(async () => {}); - - await expect(declare(registry, { setTools }, { url: 42 })).resolves.toBe( - false, - ); - expect(registry.revealDeferredTool).not.toHaveBeenCalled(); - expect(setTools).not.toHaveBeenCalled(); - }); - - it('keeps a target demoted by tools.eager behind the bridge', async () => { - const registry = { - ...revealRegistry(), - isPermissionDeferred: vi.fn(() => true), - }; - const setTools = vi.fn(async () => {}); - - await expect(declare(registry, { setTools }, {})).resolves.toBe(false); - expect(registry.revealDeferredTool).not.toHaveBeenCalled(); - expect(setTools).not.toHaveBeenCalled(); - }); - - it('does nothing for a visible target, without a client, or in a subagent', async () => { - const setTools = vi.fn(async () => {}); - const visible = revealRegistry(false); - await expect(declare(visible, { setTools }, {})).resolves.toBe(false); - - const noClient = revealRegistry(); - await expect(declare(noClient, null, {})).resolves.toBe(false); - - const subagent = revealRegistry(); - await expect( - runWithAgentContext('worker', () => declare(subagent, { setTools }, {})), - ).resolves.toBe(false); - - for (const registry of [visible, noClient, subagent]) { - expect(registry.revealDeferredTool).not.toHaveBeenCalled(); - } - expect(setTools).not.toHaveBeenCalled(); - }); - - it('keeps primary declarations unchanged in an agent chat without an agent ID', async () => { - const registry = revealRegistry(); - const setTools = vi.fn(async () => {}); - - await runWithAgentChat({} as LlmChat, async () => { - expect(getCurrentAgentId()).toBeNull(); - await expect(declare(registry, { setTools }, {})).resolves.toBe(false); - }); - - expect(registry.revealDeferredTool).not.toHaveBeenCalled(); - expect(setTools).not.toHaveBeenCalled(); - }); - - it('rolls the reveal back when the declaration refresh fails', async () => { - const registry = revealRegistry(); - const setTools = vi.fn(async () => { - throw new Error('refresh failed'); - }); - - await expect(declare(registry, { setTools }, {})).resolves.toBe(false); - expect(registry.revealDeferredTool).toHaveBeenCalledWith('web_fetch'); - expect(registry.unrevealDeferredTool).toHaveBeenCalledWith('web_fetch'); - }); - - it('tells the model to call the declared tool by name', () => { - expect(describeDirectDeclaration('web_fetch')).toContain( - 'call "web_fetch" by name', - ); - expect(describeDirectDeclaration('web_fetch')).toContain( - `not through ${ToolNames.TOOL_CALL}`, - ); - }); -}); - describe('ToolCallTool', () => { it('is an always-visible bridge with a stable generic schema', () => { const tool = new ToolCallTool(); diff --git a/packages/core/src/tools/tool-call.ts b/packages/core/src/tools/tool-call.ts index 83e3e7f6776..107812344a2 100644 --- a/packages/core/src/tools/tool-call.ts +++ b/packages/core/src/tools/tool-call.ts @@ -11,7 +11,6 @@ import type { } from './tools.js'; import { BaseDeclarativeTool, BaseToolInvocation, Kind } from './tools.js'; import { ToolErrorType } from './tool-error.js'; -import { getCurrentAgentChat } from '../agents/runtime/agent-context.js'; import { canonicalToolName, resolveRegisteredToolName, @@ -76,52 +75,6 @@ export function describeBridgedArgumentError( return `Deferred tool "${targetName}" (called through ${ToolNames.TOOL_CALL}) rejected the arguments: ${message.replace(/\.$/, '')}. Pass arguments matching the schema returned by ${ToolNames.TOOL_SEARCH} for "${targetName}".`; } -/** - * Recovery for a hidden target that a model could not reach through the - * bridge (#12889). `tool_call` declares `arguments` as an open object, so a - * model that cannot fill it — a constrained decoder, or a model that ignores - * the target schema it reviewed — keeps sending `{}` and stops on the retry - * guard. When a hidden deferred target reached through tool_call received no - * arguments at all and its own validation rejected that, declare it directly - * so the next request carries its real parameter schema. - * - * Primary session only: a subagent's declarations are not the LlmClient's. - * Narrow on purpose — an ordinary argument mistake (any non-empty arguments) - * never widens the declaration list, because a revealed tool stays declared - * for the rest of the session. The cost is one prompt-prefix rebuild, paid - * only on this failure path. Returns whether the target is now declared. - */ -export async function declareTargetAfterEmptyBridgedCall( - registry: ToolRegistry, - client: { setTools(): Promise } | null | undefined, - targetName: string, - args: Record | undefined, -): Promise { - if (args && Object.keys(args).length > 0) return false; - if (!client || getCurrentAgentChat() || isSubagentLikeExecutionContext()) - return false; - if ( - typeof registry.revealDeferredTool !== 'function' || - registry.isPermissionDeferred?.(targetName) === true || - !registry.isDeferredAndHidden(targetName) - ) { - return false; - } - registry.revealDeferredTool(targetName); - try { - await client.setTools(); - } catch { - registry.unrevealDeferredTool?.(targetName); - return false; - } - return true; -} - -/** Appended to the bridged validation error once the target is declared. */ -export function describeDirectDeclaration(targetName: string): string { - return ` "${targetName}" is now declared directly with its full parameter schema: call "${targetName}" by name with its required arguments, not through ${ToolNames.TOOL_CALL}.`; -} - export async function resolveDeferredToolCall( registry: ToolRegistry, envelope: Record, From 15aac030ea565faa8499db2a173b073c3059f29c Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Fri, 2 Oct 2026 23:42:19 +0800 Subject: [PATCH 26/27] test(acp): Verify deferred-tool validation diagnostics --- .../acp-integration/session/Session.test.ts | 223 ++++++++++-------- .../responses-converter.test.ts | 3 - .../responses-converter.ts | 12 +- 3 files changed, 127 insertions(+), 111 deletions(-) diff --git a/packages/cli/src/acp-integration/session/Session.test.ts b/packages/cli/src/acp-integration/session/Session.test.ts index 6dd6f6f85e3..164684f36a3 100644 --- a/packages/cli/src/acp-integration/session/Session.test.ts +++ b/packages/cli/src/acp-integration/session/Session.test.ts @@ -18458,109 +18458,132 @@ describe('Session', () => { }, ); - it('routes tool_call through a hidden deferred tool in ACP', async () => { - mockConfig.getApprovalMode = vi.fn().mockReturnValue(ApprovalMode.YOLO); - const execute = vi.fn().mockResolvedValue({ - llmContent: 'created issue', - returnDisplay: 'created issue', - }); - const bridge = { - name: core.ToolNames.TOOL_CALL, - kind: core.Kind.Other, - description: 'Deferred tool bridge', - build: vi.fn((params: Record) => ({ params })), - }; - // The bridge needs both halves registered: resolution rejects a - // hidden target when tool_search is unregistered (R1-5 guard). - const toolSearch = { - name: core.ToolNames.TOOL_SEARCH, - kind: core.Kind.Other, - description: 'Deferred tool discovery', - build: vi.fn((params: Record) => ({ params })), - }; - const target = { - name: 'mcp__github__create_issue', - kind: core.Kind.Other, - displayName: 'CreateIssue', - description: 'Creates an issue', - canUpdateOutput: false, - isOutputMarkdown: false, - build: vi.fn().mockImplementation((params) => ({ - params, - getDefaultPermission: vi.fn().mockResolvedValue('allow'), - getDescription: vi.fn().mockReturnValue('create issue'), - toolLocations: vi.fn().mockReturnValue([]), - execute, - })), - }; - mockToolRegistry.getTool.mockImplementation((name: string) => - name === bridge.name - ? bridge - : name === target.name - ? target - : name === toolSearch.name - ? toolSearch - : undefined, - ); - mockToolRegistry.ensureTool.mockImplementation(async (name: string) => - name === bridge.name - ? bridge - : name === target.name - ? target - : name === toolSearch.name - ? toolSearch - : undefined, - ); - mockToolRegistry.isDeferredAndHidden.mockImplementation( - (name: string) => name === target.name, - ); - const toolLoopState = { - totalToolCalls: 0, - invalidToolParamErrors: new Map(), - toolCallKeyCounts: new Map(), - maxToolCallKeyRepeat: 0, - loopDetected: false, - }; - - const result = await ( - session as unknown as { - runToolCalls: ( - abortSignal: AbortSignal, - promptId: string, - calls: FunctionCall[], - loopState: typeof toolLoopState, - ) => Promise<{ parts: Part[] }>; + it.each(['success', 'build error', 'execute error'] as const)( + 'routes tool_call through a hidden deferred tool in ACP: %s', + async (outcome) => { + mockConfig.getApprovalMode = vi + .fn() + .mockReturnValue(ApprovalMode.YOLO); + const execute = vi.fn().mockResolvedValue({ + llmContent: 'created issue', + returnDisplay: 'created issue', + }); + const bridge = { + name: core.ToolNames.TOOL_CALL, + kind: core.Kind.Other, + description: 'Deferred tool bridge', + build: vi.fn((params: Record) => ({ params })), + }; + // The bridge needs both halves registered: resolution rejects a + // hidden target when tool_search is unregistered (R1-5 guard). + const toolSearch = { + name: core.ToolNames.TOOL_SEARCH, + kind: core.Kind.Other, + description: 'Deferred tool discovery', + build: vi.fn((params: Record) => ({ params })), + }; + const target = { + name: 'mcp__github__create_issue', + kind: core.Kind.Other, + displayName: 'CreateIssue', + description: 'Creates an issue', + canUpdateOutput: false, + isOutputMarkdown: false, + build: vi.fn().mockImplementation((params) => ({ + params, + getDefaultPermission: vi.fn().mockResolvedValue('allow'), + getDescription: vi.fn().mockReturnValue('create issue'), + toolLocations: vi.fn().mockReturnValue([]), + execute, + })), + }; + if (outcome === 'build error') { + target.build.mockImplementation(() => { + throw new Error("params must have required property 'title'."); + }); + } else if (outcome === 'execute error') { + execute.mockRejectedValue(new Error('Remote service unavailable')); } - ).runToolCalls( - new AbortController().signal, - 'prompt-tool-call-bridge', - [ - { - id: 'bridge-call', - name: core.ToolNames.TOOL_CALL, - args: { - name: target.name, - arguments: { title: 'Cache-safe tools' }, + mockToolRegistry.getTool.mockImplementation((name: string) => + name === bridge.name + ? bridge + : name === target.name + ? target + : name === toolSearch.name + ? toolSearch + : undefined, + ); + mockToolRegistry.ensureTool.mockImplementation( + async (name: string) => + name === bridge.name + ? bridge + : name === target.name + ? target + : name === toolSearch.name + ? toolSearch + : undefined, + ); + mockToolRegistry.isDeferredAndHidden.mockImplementation( + (name: string) => name === target.name, + ); + const toolLoopState = { + totalToolCalls: 0, + invalidToolParamErrors: new Map(), + toolCallKeyCounts: new Map(), + maxToolCallKeyRepeat: 0, + loopDetected: false, + }; + + const result = await ( + session as unknown as { + runToolCalls: ( + abortSignal: AbortSignal, + promptId: string, + calls: FunctionCall[], + loopState: typeof toolLoopState, + ) => Promise<{ parts: Part[] }>; + } + ).runToolCalls( + new AbortController().signal, + 'prompt-tool-call-bridge', + [ + { + id: 'bridge-call', + name: core.ToolNames.TOOL_CALL, + args: { + name: target.name, + arguments: { title: 'Cache-safe tools' }, + }, }, - }, - ], - toolLoopState, - ); + ], + toolLoopState, + ); - expect(execute).toHaveBeenCalledOnce(); - expect(target.build).toHaveBeenCalledWith({ - title: 'Cache-safe tools', - }); - expect(result.parts[0]?.functionResponse).toMatchObject({ - id: 'bridge-call', - name: core.ToolNames.TOOL_CALL, - response: { output: 'created issue' }, - }); - expect(mockLlmClient.recordCompletedToolCall).toHaveBeenCalledWith( - target.name, - { title: 'Cache-safe tools' }, - ); - }); + expect(execute).toHaveBeenCalledTimes( + outcome === 'build error' ? 0 : 1, + ); + expect(target.build).toHaveBeenCalledWith({ + title: 'Cache-safe tools', + }); + expect(result.parts[0]?.functionResponse).toMatchObject({ + id: 'bridge-call', + name: core.ToolNames.TOOL_CALL, + response: + outcome === 'success' + ? { output: 'created issue' } + : { + error: + outcome === 'build error' + ? `Deferred tool "${target.name}" (called through tool_call) rejected the arguments: params must have required property 'title'. Pass arguments matching the schema returned by tool_search for "${target.name}".` + : 'Remote service unavailable', + }, + }); + expect(mockLlmClient.recordCompletedToolCall).toHaveBeenCalledWith( + target.name, + { title: 'Cache-safe tools' }, + ); + }, + ); it('marks a disabled ACP tool_call as a bridge refusal', async () => { mockConfig.getPermissionManager = vi.fn().mockReturnValue({ diff --git a/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.test.ts b/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.test.ts index 69c780e1461..c27755deaa4 100644 --- a/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.test.ts +++ b/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.test.ts @@ -1356,9 +1356,6 @@ describe('normalizeResponsesParameters', () => { }); it('keeps a nested open object open, and an explicit closed one closed', () => { - // A nested `{type:'object'}` means "any object"; `properties: {}` alone - // reads as "an empty object" to a backend that constrains decoding to - // the schema (#12889). expect( normalizeResponsesParameters({ type: 'object', diff --git a/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.ts b/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.ts index 9b548527bf0..6aac0b5da83 100644 --- a/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.ts +++ b/packages/core/src/core/openaiResponsesContentGenerator/responses-converter.ts @@ -922,14 +922,10 @@ export function convertGeminiToolsToResponsesTools( * `properties: {}` so Azure accepts them. Well-formed schemas pass through * unchanged. * - * A nested object without `properties` means "any object" (JSON Schema's - * `additionalProperties` defaults to true). With only `properties: {}` added, - * a backend that constrains decoding to the schema reads it as "an empty - * object" and can produce nothing else — which is how the deferred-tool - * bridge's `tool_call.arguments` reached a Responses model as `{}` on every - * attempt (#12889). So a nested node patched here also gets an explicit - * `additionalProperties: true` unless it already states one. The root keeps - * the bare patch: a zero-arg tool's top level really is empty. + * Explicitly preserve JSON Schema's default openness for patched nested + * objects. Whether a Responses backend interpreted the implicit default as + * closed in #12889 remains unverified. Keep the existing root-object patch + * and any explicit additionalProperties constraint unchanged. */ export function normalizeResponsesParameters( schema: Record | undefined, From 67ea7b62ab3f6cfc06497f041a8d34546a7eba73 Mon Sep 17 00:00:00 2001 From: yiliang114 <1204183885@qq.com> Date: Sun, 4 Oct 2026 02:20:52 +0800 Subject: [PATCH 27/27] test(core): preserve shared bridged validation retry budget --- .../core/src/core/coreToolScheduler.test.ts | 38 +++++++++++++++++++ 1 file changed, 38 insertions(+) diff --git a/packages/core/src/core/coreToolScheduler.test.ts b/packages/core/src/core/coreToolScheduler.test.ts index 36290577da9..5ecd89b36e2 100644 --- a/packages/core/src/core/coreToolScheduler.test.ts +++ b/packages/core/src/core/coreToolScheduler.test.ts @@ -1716,6 +1716,44 @@ describe('CoreToolScheduler', () => { expect(message).not.toContain('Deferred tool'); }); + it('shares validation retries across bridged and direct calls', async () => { + const { deferred, scheduler, onAllToolCallsComplete } = bridgeWithDeferred({ + params: URL_REQUIRED_PARAMS, + }); + + for (const [index, name] of [ + ToolNames.TOOL_CALL, + deferred.name, + ToolNames.TOOL_CALL, + ].entries()) { + onAllToolCallsComplete.mockClear(); + await scheduler.schedule( + toolRequest( + `mixed-validation-${index}`, + name, + name === ToolNames.TOOL_CALL + ? { name: deferred.name, arguments: {} } + : {}, + 'prompt-mixed-validation', + ), + new AbortController().signal, + ); + + const completed = firstBatch(onAllToolCallsComplete)[0]; + expectStatus(completed, 'error'); + expect(completed.response.errorType).toBe( + ToolErrorType.INVALID_TOOL_PARAMS, + ); + expect(functionResponseOf(completed)?.name).toBe(name); + const message = completed.response.error?.message ?? ''; + expect(message).toContain("must have required property 'url'"); + expect(message.includes('Deferred tool')).toBe( + name === ToolNames.TOOL_CALL, + ); + expect(message.includes('RETRY LOOP DETECTED')).toBe(index === 2); + } + }); + it('prunes the bridge-keyed retry counter across a successful bridged execution', async () => { // R1-18: invalid envelopes record under the model-facing name // (`tool_call:`), but a resolved envelope is renamed to the TARGET