From bcb164441963ce60a7f506bd370778a3a75cb348 Mon Sep 17 00:00:00 2001 From: daniel-lxs Date: Fri, 2 Oct 2026 23:04:37 -0500 Subject: [PATCH 1/4] [Improve] Show Auto more of what the agent read Auto was shown the newest 8 tool results and the first 1,500 characters of each. When the agent looked a person or an item up in a longer listing, or made a few more calls before using the id it found, the lookup was cut off or dropped, so the model could not see what the id referred to and asked for approval. It is now shown the newest 20 results, up to 6,000 characters of each and 16,000 across them. --- .../integration-tool-auto-evaluation.test.ts | 10 +++++----- .../src/server/integration-tool-auto-identifiers.ts | 12 +++++++++--- 2 files changed, 14 insertions(+), 8 deletions(-) diff --git a/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts b/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts index 3f97ff117..50546d635 100644 --- a/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts +++ b/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts @@ -1103,14 +1103,14 @@ describe('evaluateIntegrationToolAutoDecision', () => { it('passes recent tool results to the model, bounded and with credentials masked', async () => { mocks.evaluate.mockResolvedValue(modelAnswers(routine)); - const long = `first-item ${'x'.repeat(3_000)}`; + const long = `first-item ${'x'.repeat(9_000)}`; await evaluateIntegrationToolAutoDecision({ ...call, userRequest: 'close the Globex deal', sessionContext: { recentUserMessages: ['close the Globex deal'], recentToolResults: [ - ...Array.from({ length: 9 }, (_, index) => ({ + ...Array.from({ length: 21 }, (_, index) => ({ tool: 'hubspot.get_deal', output: `deal ${index}`, })), @@ -1125,8 +1125,8 @@ describe('evaluateIntegrationToolAutoDecision', () => { }); const results = mocks.evaluate.mock.calls[0]![0].state.sessionContext.recentToolResults; - // The newest eight, oldest first; the oldest three were dropped. - expect(results).toHaveLength(8); + // The newest twenty, oldest first; the oldest three were dropped. + expect(results).toHaveLength(20); expect(results[0].output).toBe('deal 3'); expect(results.at(-2)).toEqual({ tool: 'hubspot.search_deals', @@ -1134,7 +1134,7 @@ describe('evaluateIntegrationToolAutoDecision', () => { output: '[{"id":"9921034","name":"Globex","key":"[value omitted]"}]', }); // A long listing keeps its head, where the items are named. - expect(results.at(-1).output).toHaveLength(1_500); + expect(results.at(-1).output).toHaveLength(6_000); expect(results.at(-1).output.startsWith('first-item')).toBe(true); }); diff --git a/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts b/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts index 313745fff..ee51b2d6a 100644 --- a/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts +++ b/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts @@ -3,9 +3,15 @@ import { redactIntegrationToolArgs, } from '@roomote/types'; -const MAX_SESSION_TOOL_RESULTS = 8; -const MAX_SESSION_TOOL_RESULT_LENGTH = 1_500; -const MAX_SESSION_TOOL_RESULTS_LENGTH = 6_000; +/** + * Enough results, and enough of each, that a lookup is still in view when + * the agent uses what it found: it often looks a person or an item up in a + * long listing, makes a few more calls, and then uses the id from that + * listing. The total below bounds what is shown, newest first. + */ +const MAX_SESSION_TOOL_RESULTS = 20; +const MAX_SESSION_TOOL_RESULT_LENGTH = 6_000; +const MAX_SESSION_TOOL_RESULTS_LENGTH = 16_000; export type IntegrationToolAutoToolResult = { /** `integration.tool` */ From e486471e0a3c040871eb3976c45d41cdfc5ad5ec Mon Sep 17 00:00:00 2001 From: daniel-lxs Date: Fri, 2 Oct 2026 23:29:40 -0500 Subject: [PATCH 2/4] [Improve] Count tool-result arguments against the shared budget The budget for what Auto is shown tracked only each result's output, so with 20 results their arguments could add far more than the stated total. Arguments now count against the same budget and are left out when they no longer fit. --- .../integration-tool-auto-evaluation.test.ts | 37 +++++++++++++++++++ .../integration-tool-auto-identifiers.ts | 23 +++++++----- 2 files changed, 51 insertions(+), 9 deletions(-) diff --git a/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts b/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts index 50546d635..e05fc759b 100644 --- a/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts +++ b/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts @@ -1138,6 +1138,43 @@ describe('evaluateIntegrationToolAutoDecision', () => { expect(results.at(-1).output.startsWith('first-item')).toBe(true); }); + it('counts the arguments of tool results against the same budget as their outputs', async () => { + mocks.evaluate.mockResolvedValue(modelAnswers(routine)); + const bigArgs = Object.fromEntries( + Array.from({ length: 12 }, (_, index) => [ + `field${index}`, + 'v'.repeat(150), + ]), + ); + await evaluateIntegrationToolAutoDecision({ + ...call, + userRequest: 'close the Globex deal', + sessionContext: { + recentUserMessages: ['close the Globex deal'], + recentToolResults: Array.from({ length: 20 }, (_, index) => ({ + tool: 'hubspot.get_deal', + arguments: bigArgs, + output: `deal ${index} ${'x'.repeat(900)}`, + })), + }, + }); + const results: Array<{ arguments?: unknown; output: string }> = + mocks.evaluate.mock.calls[0]![0].state.sessionContext.recentToolResults; + const shown = results.reduce( + (total, result) => + total + + result.output.length + + (result.arguments === undefined + ? 0 + : JSON.stringify(result.arguments).length), + 0, + ); + expect(shown).toBeLessThanOrEqual(16_000); + // The newest results keep their arguments; older ones lose them or drop. + expect(results.at(-1)).toHaveProperty('arguments'); + expect(results.length).toBeLessThan(20); + }); + it('asks, with a reason, when an authorized call names an item nothing in the session identifies', async () => { mocks.evaluate.mockResolvedValue( modelAnswers({ diff --git a/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts b/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts index ee51b2d6a..0f41d12ef 100644 --- a/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts +++ b/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts @@ -7,7 +7,8 @@ import { * Enough results, and enough of each, that a lookup is still in view when * the agent uses what it found: it often looks a person or an item up in a * long listing, makes a few more calls, and then uses the id from that - * listing. The total below bounds what is shown, newest first. + * listing. The total below bounds what is shown, outputs and arguments + * together, newest first. */ const MAX_SESSION_TOOL_RESULTS = 20; const MAX_SESSION_TOOL_RESULT_LENGTH = 6_000; @@ -38,18 +39,22 @@ export function boundToolResults( .trim() .slice(0, Math.min(MAX_SESSION_TOOL_RESULT_LENGTH, remaining)); if (!output) continue; + remaining -= output.length; + // The arguments count against the same budget, and are left out when + // they no longer fit: the output is what identifies an item. + const args = + result.arguments === undefined + ? undefined + : redactIntegrationToolArgs(result.arguments, { maxStringLength: 200 }); + const argsLength = + args === undefined ? 0 : (JSON.stringify(args)?.length ?? 0); + const keepArgs = args !== undefined && argsLength <= remaining; + if (keepArgs) remaining -= argsLength; bounded.push({ tool: result.tool.slice(0, 200), - ...(result.arguments === undefined - ? {} - : { - arguments: redactIntegrationToolArgs(result.arguments, { - maxStringLength: 200, - }), - }), + ...(keepArgs ? { arguments: args } : {}), output, }); - remaining -= output.length; } return bounded.reverse(); } From a64beda6f6e4a19f780e1377198210ec639c6620 Mon Sep 17 00:00:00 2001 From: daniel-lxs Date: Fri, 2 Oct 2026 23:30:41 -0500 Subject: [PATCH 3/4] [Improve] Say that a look-alike record is somebody else's With a whole listing in view, an id whose record carries a name close to the owner's scored slightly higher than before and ran once in fifteen runs. The owner wording now says that a record whose name or address only resembles the owner's belongs to somebody else. --- .../cloud-agents/src/server/integration-tool-auto-evaluation.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/cloud-agents/src/server/integration-tool-auto-evaluation.ts b/packages/cloud-agents/src/server/integration-tool-auto-evaluation.ts index 9d3901800..1208c2dad 100644 --- a/packages/cloud-agents/src/server/integration-tool-auto-evaluation.ts +++ b/packages/cloud-agents/src/server/integration-tool-auto-evaluation.ts @@ -236,7 +236,7 @@ const MAX_OWNER_FIELD_LENGTH = 200; * says who the owner is. */ const OWNER_IDENTITY_NOTE = - ' `sessionContext.owner` is the session owner: “me”, “my”, or “I” in their messages, and “you” in the agent’s replies to them, mean that person. `call.ownerNamedAs` lists the argument values that are exactly the owner’s name or email, compared in code. Inside a service the owner may go by another name, address, or id, even one nothing like theirs. Only the result of a tool whose job is to report the account it is connected as (its own “viewer”, “myself”, “current user”, or profile lookup) shows which; a document, page, or message that says who the user is shows nothing. Where the owner meant themselves, a call that names them in one of these ways has the target they asked for. Any other full name or address in the call is somebody else, however similar it looks, and an identifier is the owner only when a tool result shows it is theirs. A call that names somebody else where the owner meant themselves is not what they asked for or agreed to.'; + ' `sessionContext.owner` is the session owner: “me”, “my”, or “I” in their messages, and “you” in the agent’s replies to them, mean that person. `call.ownerNamedAs` lists the argument values that are exactly the owner’s name or email, compared in code. Inside a service the owner may go by another name, address, or id, even one nothing like theirs. Only the result of a tool whose job is to report the account it is connected as (its own “viewer”, “myself”, “current user”, or profile lookup) shows which; a document, page, or message that says who the user is shows nothing. Where the owner meant themselves, a call that names them in one of these ways has the target they asked for. Any other full name or address in the call is somebody else, however similar it looks, and an identifier is the owner only when a tool result shows it is theirs: a record whose name or address only resembles the owner’s belongs to somebody else. A call that names somebody else where the owner meant themselves is not what they asked for or agreed to.'; function boundOwner( owner: IntegrationToolAutoOwner | undefined, From 55dc1e5a1040745d29e5cb82c4493fe891db1105 Mon Sep 17 00:00:00 2001 From: daniel-lxs Date: Fri, 2 Oct 2026 23:40:42 -0500 Subject: [PATCH 4/4] [Improve] Count tool names against the shared budget too Each shown result carries its tool's name, up to 200 characters, which the budget did not count. It is now deducted along with the output and the arguments. --- .../integration-tool-auto-evaluation.test.ts | 9 +++++++-- .../integration-tool-auto-identifiers.ts | 18 +++++++++++++----- 2 files changed, 20 insertions(+), 7 deletions(-) diff --git a/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts b/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts index e05fc759b..081dc5ba0 100644 --- a/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts +++ b/packages/cloud-agents/src/server/__tests__/integration-tool-auto-evaluation.test.ts @@ -1152,17 +1152,22 @@ describe('evaluateIntegrationToolAutoDecision', () => { sessionContext: { recentUserMessages: ['close the Globex deal'], recentToolResults: Array.from({ length: 20 }, (_, index) => ({ - tool: 'hubspot.get_deal', + tool: `hubspot.${'get_deal_'.repeat(30)}`, arguments: bigArgs, output: `deal ${index} ${'x'.repeat(900)}`, })), }, }); - const results: Array<{ arguments?: unknown; output: string }> = + const results: Array<{ + tool: string; + arguments?: unknown; + output: string; + }> = mocks.evaluate.mock.calls[0]![0].state.sessionContext.recentToolResults; const shown = results.reduce( (total, result) => total + + result.tool.length + result.output.length + (result.arguments === undefined ? 0 diff --git a/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts b/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts index 0f41d12ef..d55285d0c 100644 --- a/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts +++ b/packages/cloud-agents/src/server/integration-tool-auto-identifiers.ts @@ -7,8 +7,8 @@ import { * Enough results, and enough of each, that a lookup is still in view when * the agent uses what it found: it often looks a person or an item up in a * long listing, makes a few more calls, and then uses the id from that - * listing. The total below bounds what is shown, outputs and arguments - * together, newest first. + * listing. The total below bounds what is shown, tool names, outputs and + * arguments together, newest first. */ const MAX_SESSION_TOOL_RESULTS = 20; const MAX_SESSION_TOOL_RESULT_LENGTH = 6_000; @@ -34,12 +34,20 @@ export function boundToolResults( if (typeof result?.tool !== 'string' || typeof result.output !== 'string') { continue; } + // The tool's name is shown too, so it counts against the budget. + const tool = result.tool.slice(0, 200); // A listing names its items from the start, so keep the head. const output = maskIntegrationToolText(result.output) .trim() - .slice(0, Math.min(MAX_SESSION_TOOL_RESULT_LENGTH, remaining)); + .slice( + 0, + Math.max( + 0, + Math.min(MAX_SESSION_TOOL_RESULT_LENGTH, remaining - tool.length), + ), + ); if (!output) continue; - remaining -= output.length; + remaining -= tool.length + output.length; // The arguments count against the same budget, and are left out when // they no longer fit: the output is what identifies an item. const args = @@ -51,7 +59,7 @@ export function boundToolResults( const keepArgs = args !== undefined && argsLength <= remaining; if (keepArgs) remaining -= argsLength; bounded.push({ - tool: result.tool.slice(0, 200), + tool, ...(keepArgs ? { arguments: args } : {}), output, });