diff --git a/command-snapshot.json b/command-snapshot.json index 935c24d9..25d953e5 100644 --- a/command-snapshot.json +++ b/command-snapshot.json @@ -287,6 +287,57 @@ ], "plugin": "@salesforce/plugin-agent" }, + { + "alias": [], + "command": "agent:optimize:accept", + "flagAliases": [], + "flagChars": ["i", "o"], + "flags": ["api-version", "execution-id", "flags-dir", "json", "target-org"], + "plugin": "@salesforce/plugin-agent" + }, + { + "alias": [], + "command": "agent:optimize:results", + "flagAliases": [], + "flagChars": ["i", "o"], + "flags": ["api-version", "execution-id", "flags-dir", "json", "target-org"], + "plugin": "@salesforce/plugin-agent" + }, + { + "alias": [], + "command": "agent:optimize:run", + "flagAliases": [], + "flagChars": ["i", "o", "s"], + "flags": ["api-version", "authoring-bundle", "flags-dir", "iterations", "json", "spec", "target-org"], + "plugin": "@salesforce/plugin-agent" + }, + { + "alias": [], + "command": "agent:optimize:start", + "flagAliases": [], + "flagChars": ["b", "c", "o", "w"], + "flags": [ + "api-version", + "authoring-bundle", + "criteria", + "flags-dir", + "json", + "max-iterations", + "target-org", + "target-score", + "test-cases", + "wait" + ], + "plugin": "@salesforce/plugin-agent" + }, + { + "alias": [], + "command": "agent:optimize:status", + "flagAliases": [], + "flagChars": ["i", "o"], + "flags": ["api-version", "execution-id", "flags-dir", "json", "target-org"], + "plugin": "@salesforce/plugin-agent" + }, { "alias": [], "command": "agent:preview", diff --git a/messages/agent.optimize.accept.md b/messages/agent.optimize.accept.md new file mode 100644 index 00000000..31bca3dc --- /dev/null +++ b/messages/agent.optimize.accept.md @@ -0,0 +1,21 @@ +# summary + +Accept and publish the results of a completed HEPO optimization. + +# description + +Sends a publish signal to the optimization workflow, deploying the best-performing agent configuration as a draft. + +# examples + +- Accept and publish an optimization result: + + <%= config.bin %> <%= command.id %> --execution-id hepo:wf-abc123:run-xyz --target-org myOrg + +# flags.execution-id.summary + +Execution ID returned by `sf agent optimize start`. + +# error.acceptFailed + +Failed to accept optimization: %s diff --git a/messages/agent.optimize.results.md b/messages/agent.optimize.results.md new file mode 100644 index 00000000..0ea79e42 --- /dev/null +++ b/messages/agent.optimize.results.md @@ -0,0 +1,25 @@ +# summary + +Get the final results of a completed HEPO optimization run. + +# description + +Returns detailed results of a completed optimization, including iteration history, score improvements, and the best agent snapshot. + +# examples + +- Get results of a completed optimization: + + <%= config.bin %> <%= command.id %> --execution-id hepo:wf-abc123:run-xyz --target-org myOrg + +# flags.execution-id.summary + +Execution ID returned by `sf agent optimize start`. + +# error.resultsFailed + +Failed to get optimization results: %s + +# error.stillRunning + +Optimization %s is still running. Use `sf agent optimize status` to check progress. diff --git a/messages/agent.optimize.run.md b/messages/agent.optimize.run.md new file mode 100644 index 00000000..cf7977f9 --- /dev/null +++ b/messages/agent.optimize.run.md @@ -0,0 +1,95 @@ +# summary + +Optimize an Agentforce agent by iteratively improving its instructions. + +# description + +Runs an optimization loop that evaluates the agent against test cases, uses an LLM to propose instruction improvements, and keeps changes that improve the score. + +Each iteration: + +1. Sends test utterances to the agent via preview +2. Scores responses against expected outputs +3. Proposes instruction edits via an LLM +4. Applies edits and re-evaluates +5. Keeps improvements, rejects regressions + +Uses the org's Einstein LLM to propose improvements. Falls back to ANTHROPIC_API_KEY or OPENAI_API_KEY from the environment when Einstein is not available on the org. + +# flags.authoring-bundle.summary + +Name of the authoring bundle to optimize. + +# flags.spec.summary + +Path to optimization spec file (JSON or YAML). Contains test cases with utterances and expected response keywords. + +# flags.iterations.summary + +Number of optimization iterations to run (default: 3). + +# examples + +- Optimize an agent with 3 iterations: + + <%= config.bin %> <%= command.id %> --authoring-bundle MyAgent --spec test-cases.json --target-org my-org + +- Run 5 optimization iterations: + + <%= config.bin %> <%= command.id %> --authoring-bundle MyAgent --spec test-cases.json --target-org my-org --iterations 5 + +# output.baseline + +Baseline score: %s (%s/%s test cases passed) + +# output.iterationKeep + +Iteration %s: KEEP — score improved %s → %s (%s) + +# output.iterationReject + +Iteration %s: REJECT — score %s did not improve over %s (%s) + +# output.summary + +Optimization complete. Baseline: %s → Best: %s (%s iterations, %s kept) + +# output.noImprovement + +No improvements found after %s iterations. Agent unchanged. + +# output.alreadyPerfect + +All test cases already passing. Agent is fully optimized — no changes needed. + +# output.agentUpdated + +Agent file updated: %s + +# output.agentRestored + +Agent file restored to best version. + +# error.specNotFound + +Spec file not found: %s. + +# error.invalidSpec + +Invalid optimization spec: %s. Expected JSON with a "test_cases" array. + +# error.bundleNotFound + +Authoring bundle '%s' not found in the project. + +# error.agentFileNotFound + +Agent file not found in bundle '%s'. Expected .agent inside the authoring bundle directory. + +# error.llmCallFailed + +Einstein LLM call failed: %s. + +# error.previewFailed + +Preview session failed: %s. diff --git a/messages/agent.optimize.start.md b/messages/agent.optimize.start.md new file mode 100644 index 00000000..4dfe3431 --- /dev/null +++ b/messages/agent.optimize.start.md @@ -0,0 +1,83 @@ +# summary + +Start a server-side HEPO optimization for an Agentforce agent. + +# description + +Starts an iterative optimization workflow on the server. The workflow evaluates the agent against test cases, proposes instruction improvements via an LLM, and keeps changes that improve the score. + +Unlike `sf agent optimize run` (which runs optimization client-side), this command offloads all work to the HEPO service and only polls for progress. + +# examples + +- Start optimization and wait for completion: + + <%= config.bin %> <%= command.id %> --authoring-bundle ShoppingAgent --criteria criteria.yaml --target-org myOrg --wait 60 + +- Start optimization without waiting (check status later): + + <%= config.bin %> <%= command.id %> --authoring-bundle ShoppingAgent --criteria criteria.yaml --target-org myOrg + +- Start with custom iteration settings: + + <%= config.bin %> <%= command.id %> --authoring-bundle ShoppingAgent --criteria criteria.yaml --max-iterations 15 --target-score 0.95 --target-org myOrg + +# flags.authoring-bundle.summary + +Name of the agent authoring bundle to optimize. + +# flags.criteria.summary + +Path to criteria YAML file defining scoring metrics, gates, and weights. + +# flags.test-cases.summary + +Path to test cases YAML file. If not provided, test cases from the criteria file are used. + +# flags.max-iterations.summary + +Maximum number of optimization iterations (default: 10). + +# flags.target-score.summary + +Target composite score to stop early (default: 1.0). + +# flags.wait.summary + +Minutes to wait for completion. + +# flags.wait.description + +Poll for status updates until the optimization completes or times out. Without --wait, the command returns immediately after starting. + +# output.started + +Optimization started. Execution ID: %s + +# output.progress + +Iteration %s/%s — Baseline: %s Current: %s Best: %s + +# output.completed + +Optimization completed in %s iterations. Baseline: %s → Best: %s + +# output.timeout + +Optimization still running after %s minutes. Use `sf agent optimize status --execution-id %s` to check progress. + +# error.startFailed + +Failed to start optimization: %s + +# error.pollFailed + +Failed to poll optimization status: %s + +# error.criteriaNotFound + +Criteria file not found: %s + +# error.invalidCriteria + +Invalid criteria file: %s diff --git a/messages/agent.optimize.status.md b/messages/agent.optimize.status.md new file mode 100644 index 00000000..36d34ca6 --- /dev/null +++ b/messages/agent.optimize.status.md @@ -0,0 +1,21 @@ +# summary + +Get the status of a server-side HEPO optimization run. + +# description + +Returns the current status of a running or completed HEPO optimization, including iteration progress and score improvements. + +# examples + +- Get status of an optimization run: + + <%= config.bin %> <%= command.id %> --execution-id hepo:wf-abc123:run-xyz --target-org myOrg + +# flags.execution-id.summary + +Execution ID returned by `sf agent optimize start`. + +# error.statusFailed + +Failed to get optimization status: %s diff --git a/package.json b/package.json index e3dcb5fc..d8af35d5 100644 --- a/package.json +++ b/package.json @@ -91,6 +91,10 @@ "description": "Commands to generate agent artifacts, such as the agent spec YAML file, authoring bundle, and test spec file.", "external": true }, + "optimize": { + "description": "Commands to optimize agents.", + "external": true + }, "validate": { "description": "Command to validate an Agent Script file.", "external": true @@ -258,5 +262,6 @@ } }, "exports": "./lib/index.js", - "type": "module" + "type": "module", + "packageManager": "yarn@1.22.22+sha512.a6b2f7906b721bba3d67d4aff083df04dad64c399707841b7acf00f6b133b7ac24255f2652fa22ae3534329dc6180534e98d17432037ff6fd140556e2bb3137e" } diff --git a/schemas/agent-optimize-accept.json b/schemas/agent-optimize-accept.json new file mode 100644 index 00000000..d97d648a --- /dev/null +++ b/schemas/agent-optimize-accept.json @@ -0,0 +1,25 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "$ref": "#/definitions/AgentOptimizeAcceptResult", + "definitions": { + "AgentOptimizeAcceptResult": { + "$ref": "#/definitions/OptimizationAcceptResult" + }, + "OptimizationAcceptResult": { + "type": "object", + "properties": { + "executionId": { + "type": "string" + }, + "published": { + "type": "boolean" + }, + "message": { + "type": "string" + } + }, + "required": ["executionId", "published", "message"], + "additionalProperties": false + } + } +} diff --git a/schemas/agent-optimize-results.json b/schemas/agent-optimize-results.json new file mode 100644 index 00000000..11ded123 --- /dev/null +++ b/schemas/agent-optimize-results.json @@ -0,0 +1,68 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "$ref": "#/definitions/AgentOptimizeResultsResult", + "definitions": { + "AgentOptimizeResultsResult": { + "$ref": "#/definitions/OptimizationResults" + }, + "OptimizationResults": { + "type": "object", + "properties": { + "executionId": { + "type": "string" + }, + "status": { + "type": "string" + }, + "iterationsRun": { + "type": "number" + }, + "baselineScore": { + "type": "number" + }, + "finalScore": { + "type": "number" + }, + "bestScore": { + "type": "number" + }, + "bestIteration": { + "type": "number" + }, + "iterationHistory": { + "type": "array", + "items": { + "$ref": "#/definitions/IterationEntry" + } + } + }, + "required": [ + "executionId", + "status", + "iterationsRun", + "baselineScore", + "finalScore", + "bestScore", + "bestIteration", + "iterationHistory" + ], + "additionalProperties": false + }, + "IterationEntry": { + "type": "object", + "properties": { + "iteration": { + "type": "number" + }, + "compositeScore": { + "type": "number" + }, + "gatesPassed": { + "type": "boolean" + } + }, + "required": ["iteration", "compositeScore", "gatesPassed"], + "additionalProperties": false + } + } +} diff --git a/schemas/agent-optimize-run.json b/schemas/agent-optimize-run.json new file mode 100644 index 00000000..571897ef --- /dev/null +++ b/schemas/agent-optimize-run.json @@ -0,0 +1,51 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "$ref": "#/definitions/OptimizeRunResult", + "definitions": { + "OptimizeRunResult": { + "type": "object", + "properties": { + "baselineScore": { + "type": "number" + }, + "bestScore": { + "type": "number" + }, + "iterations": { + "type": "array", + "items": { + "type": "object", + "properties": { + "iteration": { + "type": "number" + }, + "mutation": { + "type": "string" + }, + "score": { + "type": "number" + }, + "passCount": { + "type": "number" + }, + "totalCount": { + "type": "number" + }, + "decision": { + "type": "string", + "enum": ["KEEP", "REJECT"] + } + }, + "required": ["iteration", "mutation", "score", "passCount", "totalCount", "decision"], + "additionalProperties": false + } + }, + "agentFile": { + "type": "string" + } + }, + "required": ["baselineScore", "bestScore", "iterations", "agentFile"], + "additionalProperties": false + } + } +} diff --git a/schemas/agent-optimize-start.json b/schemas/agent-optimize-start.json new file mode 100644 index 00000000..23d833e3 --- /dev/null +++ b/schemas/agent-optimize-start.json @@ -0,0 +1,28 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "$ref": "#/definitions/AgentOptimizeStartResult", + "definitions": { + "AgentOptimizeStartResult": { + "type": "object", + "properties": { + "executionId": { + "type": "string" + }, + "status": { + "type": "string" + }, + "baselineScore": { + "type": "number" + }, + "bestScore": { + "type": "number" + }, + "iterationsRun": { + "type": "number" + } + }, + "required": ["executionId", "status"], + "additionalProperties": false + } + } +} diff --git a/schemas/agent-optimize-status.json b/schemas/agent-optimize-status.json new file mode 100644 index 00000000..109776cf --- /dev/null +++ b/schemas/agent-optimize-status.json @@ -0,0 +1,57 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "$ref": "#/definitions/AgentOptimizeStatusResult", + "definitions": { + "AgentOptimizeStatusResult": { + "$ref": "#/definitions/OptimizationStatus" + }, + "OptimizationStatus": { + "type": "object", + "properties": { + "executionId": { + "type": "string" + }, + "status": { + "$ref": "#/definitions/OptimizationStatusEnum" + }, + "currentIteration": { + "type": "number" + }, + "maxIterations": { + "type": "number" + }, + "baselineScore": { + "type": "number" + }, + "currentScore": { + "type": "number" + }, + "bestScore": { + "type": "number" + }, + "bestIteration": { + "type": "number" + }, + "message": { + "type": "string" + } + }, + "required": [ + "executionId", + "status", + "currentIteration", + "maxIterations", + "baselineScore", + "currentScore", + "bestScore", + "bestIteration", + "message" + ], + "additionalProperties": false + }, + "OptimizationStatusEnum": { + "type": "string", + "enum": ["RUNNING", "COMPLETED", "FAILED", "STOPPED_EARLY"] + } + } +} diff --git a/src/commands/agent/optimize/accept.ts b/src/commands/agent/optimize/accept.ts new file mode 100644 index 00000000..042e658d --- /dev/null +++ b/src/commands/agent/optimize/accept.ts @@ -0,0 +1,61 @@ +/* + * Copyright 2026, Salesforce, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +import { SfCommand, Flags } from '@salesforce/sf-plugins-core'; +import { Messages, SfError } from '@salesforce/core'; +import { AgentOptimization, type OptimizationAcceptResult } from '@salesforce/agents'; + +Messages.importMessagesDirectoryFromMetaUrl(import.meta.url); +const messages = Messages.loadMessages('@salesforce/plugin-agent', 'agent.optimize.accept'); + +export type AgentOptimizeAcceptResult = OptimizationAcceptResult; + +export default class AgentOptimizeAccept extends SfCommand { + public static readonly summary = messages.getMessage('summary'); + public static readonly description = messages.getMessage('description'); + public static readonly examples = messages.getMessages('examples'); + public static readonly enableJsonFlag = true; + + public static readonly flags = { + 'target-org': Flags.requiredOrg(), + 'api-version': Flags.orgApiVersion(), + 'execution-id': Flags.string({ + summary: messages.getMessage('flags.execution-id.summary'), + char: 'i', + required: true, + }), + }; + + public async run(): Promise { + const { flags } = await this.parse(AgentOptimizeAccept); + const connection = flags['target-org'].getConnection(flags['api-version']); + + let result: OptimizationAcceptResult; + try { + result = await AgentOptimization.accept(connection, flags['execution-id']); + } catch (error) { + const wrapped = SfError.wrap(error); + throw new SfError(messages.getMessage('error.acceptFailed', [wrapped.message]), 'AcceptFailed', [], 4, wrapped); + } + + if (result.published) { + this.log(`Optimization ${result.executionId} accepted and published.`); + } else { + this.log(`Optimization ${result.executionId}: ${result.message}`); + } + + return result; + } +} diff --git a/src/commands/agent/optimize/results.ts b/src/commands/agent/optimize/results.ts new file mode 100644 index 00000000..4de80eb0 --- /dev/null +++ b/src/commands/agent/optimize/results.ts @@ -0,0 +1,83 @@ +/* + * Copyright 2026, Salesforce, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +import { SfCommand, Flags } from '@salesforce/sf-plugins-core'; +import { Messages, SfError } from '@salesforce/core'; +import { AgentOptimization, type OptimizationResults } from '@salesforce/agents'; + +Messages.importMessagesDirectoryFromMetaUrl(import.meta.url); +const messages = Messages.loadMessages('@salesforce/plugin-agent', 'agent.optimize.results'); + +export type AgentOptimizeResultsResult = OptimizationResults; + +export default class AgentOptimizeResults extends SfCommand { + public static readonly summary = messages.getMessage('summary'); + public static readonly description = messages.getMessage('description'); + public static readonly examples = messages.getMessages('examples'); + public static readonly enableJsonFlag = true; + + public static readonly flags = { + 'target-org': Flags.requiredOrg(), + 'api-version': Flags.orgApiVersion(), + 'execution-id': Flags.string({ + summary: messages.getMessage('flags.execution-id.summary'), + char: 'i', + required: true, + }), + }; + + public async run(): Promise { + const { flags } = await this.parse(AgentOptimizeResults); + const connection = flags['target-org'].getConnection(flags['api-version']); + + let result: OptimizationResults; + try { + result = await AgentOptimization.results(connection, flags['execution-id']); + } catch (error) { + const wrapped = SfError.wrap(error); + if (wrapped.message.includes('still running')) { + throw new SfError( + messages.getMessage('error.stillRunning', [flags['execution-id']]), + 'StillRunning', + ['Use `sf agent optimize status` to check progress.'], + 4, + wrapped + ); + } + throw new SfError(messages.getMessage('error.resultsFailed', [wrapped.message]), 'ResultsFailed', [], 4, wrapped); + } + + this.log(`Execution: ${result.executionId}`); + this.log(`Status: ${result.status}`); + this.log(`Iterations: ${result.iterationsRun}`); + this.log( + `Baseline: ${result.baselineScore.toFixed(2)} → Final: ${result.finalScore.toFixed( + 2 + )} (Best: ${result.bestScore.toFixed(2)} at iteration ${result.bestIteration})` + ); + + if (result.iterationHistory?.length) { + this.log('\nIteration History:'); + this.log(' # Score Gates'); + this.log(' --- ------- -----'); + for (const iter of result.iterationHistory) { + const gates = iter.gatesPassed ? 'PASS' : 'FAIL'; + this.log(` ${String(iter.iteration).padStart(3)} ${iter.compositeScore.toFixed(3).padStart(7)} ${gates}`); + } + } + + return result; + } +} diff --git a/src/commands/agent/optimize/run.ts b/src/commands/agent/optimize/run.ts new file mode 100644 index 00000000..86aa55e6 --- /dev/null +++ b/src/commands/agent/optimize/run.ts @@ -0,0 +1,532 @@ +/* + * Copyright 2026, Salesforce, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +import { readFile, writeFile } from 'node:fs/promises'; +import { join } from 'node:path'; +import { Flags, SfCommand, toHelpSection } from '@salesforce/sf-plugins-core'; +import { Connection, EnvironmentVariable, Messages, SfError, SfProject } from '@salesforce/core'; +import { Agent, findAuthoringBundle } from '@salesforce/agents'; +import { parse as parseYaml } from 'yaml'; + +Messages.importMessagesDirectoryFromMetaUrl(import.meta.url); +const messages = Messages.loadMessages('@salesforce/plugin-agent', 'agent.optimize.run'); + +// --------------------------------------------------------------------------- +// Types +// --------------------------------------------------------------------------- + +type TestCase = { + utterance: string; + expected_response_contains?: string[]; + expected_topic?: string; +}; + +type OptimizationSpec = { + test_cases: TestCase[]; +}; + +type TestCaseResult = { + utterance: string; + response: string; + passed: boolean; + matchedKeywords: string[]; + missedKeywords: string[]; +}; + +type IterationResult = { + iteration: number; + mutation: string; + score: number; + passCount: number; + totalCount: number; + decision: 'KEEP' | 'REJECT'; +}; + +export type OptimizeRunResult = { + baselineScore: number; + bestScore: number; + iterations: IterationResult[]; + agentFile: string; +}; + +// --------------------------------------------------------------------------- +// LLM client — tries Einstein Prompt Generations API first, then falls back +// to ANTHROPIC_API_KEY or OPENAI_API_KEY from the environment. +// --------------------------------------------------------------------------- + +type PromptGenerationResponse = { + generations: Array<{ text: string; responseId: string }>; +}; + +type AnthropicResponse = { + content: Array<{ type: string; text: string }>; +}; + +type OpenAIResponse = { + choices: Array<{ message: { content: string } }>; +}; + +async function callLlmEinstein(conn: Connection, prompt: string): Promise { + const body = { + isPreview: false, + inputParams: { + valueMap: { + 'Input:prompt_text': { + value: prompt, + }, + }, + }, + additionalConfig: { + applicationName: 'PromptTemplateGenerationsInvocable', + maxTokens: 4096, + temperature: 0, + }, + }; + + const url = `/services/data/v${conn.version}/einstein/prompt-templates/AgentOptimizer/generations`; + + const resp = await conn.request({ + method: 'POST', + url, + body: JSON.stringify(body), + }); + + if (!resp.generations?.length) { + throw new Error('Empty response from Einstein LLM'); + } + return resp.generations[0].text; +} + +async function callLlmAnthropic(apiKey: string, prompt: string): Promise { + const resp = await fetch('https://api.anthropic.com/v1/messages', { + method: 'POST', + headers: { + 'x-api-key': apiKey, + 'anthropic-version': '2023-06-01', + 'content-type': 'application/json', + }, + body: JSON.stringify({ + model: 'claude-sonnet-4-20250514', + // eslint-disable-next-line camelcase + max_tokens: 4096, + messages: [{ role: 'user', content: prompt }], + }), + }); + if (!resp.ok) throw new Error(`Anthropic API error: ${resp.status} ${resp.statusText}`); + const data = (await resp.json()) as AnthropicResponse; + return data.content[0].text; +} + +async function callLlmOpenAI(apiKey: string, prompt: string): Promise { + const resp = await fetch('https://api.openai.com/v1/chat/completions', { + method: 'POST', + headers: { + Authorization: `Bearer ${apiKey}`, + 'Content-Type': 'application/json', + }, + body: JSON.stringify({ + model: 'gpt-4o', + // eslint-disable-next-line camelcase + max_tokens: 4096, + temperature: 0, + messages: [{ role: 'user', content: prompt }], + }), + }); + if (!resp.ok) throw new Error(`OpenAI API error: ${resp.status} ${resp.statusText}`); + const data = (await resp.json()) as OpenAIResponse; + return data.choices[0].message.content; +} + +async function callLlm(conn: Connection, prompt: string): Promise { + // Try Einstein first + try { + return await callLlmEinstein(conn, prompt); + } catch { + // Einstein not available — fall through to external providers + } + + // Fallback: ANTHROPIC_API_KEY + const anthropicKey = process.env.ANTHROPIC_API_KEY; + if (anthropicKey) { + return callLlmAnthropic(anthropicKey, prompt); + } + + // Fallback: OPENAI_API_KEY + const openaiKey = process.env.OPENAI_API_KEY; + if (openaiKey) { + return callLlmOpenAI(openaiKey, prompt); + } + + throw new SfError(messages.getMessage('error.llmCallFailed', ['No LLM provider available']), 'LlmProviderError', [ + 'Einstein Prompt Generations API is not available on this org.', + 'Set ANTHROPIC_API_KEY or OPENAI_API_KEY environment variable as a fallback.', + 'Or enable Einstein generative AI and assign EinsteinGPTPromptTemplateUser permission set.', + ]); +} + +// --------------------------------------------------------------------------- +// Agent file helpers +// --------------------------------------------------------------------------- + +function locateAgentFile(project: SfProject, bundleName: string): string { + const dirs = project.getPackageDirectories().map((d) => d.fullPath); + const bundleDir = findAuthoringBundle(dirs, bundleName); + if (!bundleDir) { + throw new SfError(messages.getMessage('error.bundleNotFound', [bundleName])); + } + return join(bundleDir, `${bundleName}.agent`); +} + +// --------------------------------------------------------------------------- +// Scoring +// --------------------------------------------------------------------------- + +async function evaluateAgent( + conn: Connection, + project: SfProject, + bundleName: string, + testCases: TestCase[], + log: (msg: string) => void +): Promise<{ score: number; passCount: number; results: TestCaseResult[] }> { + const results: TestCaseResult[] = []; + + // Init once — Agent.init with aabName compiles the .agent file and returns a ScriptAgent + const agent = await Agent.init({ + connection: conn, + project, + aabName: bundleName, + }); + agent.preview.setMockMode('Mock'); + + for (const tc of testCases) { + let responseText = ''; + try { + // Start a fresh session per utterance to avoid conversational state leaking + // eslint-disable-next-line no-await-in-loop + await agent.preview.start({}); + // eslint-disable-next-line no-await-in-loop + const response = await agent.preview.send(tc.utterance); + responseText = response.messages + .filter((m) => m.type === 'Inform') + .map((m) => m.message ?? '') + .join(' '); + // eslint-disable-next-line no-await-in-loop + await agent.preview.end(); + } catch (e) { + log(` Preview failed for "${tc.utterance}": ${(e as Error).message}`); + try { + // eslint-disable-next-line no-await-in-loop + await agent.preview.end(); + } catch { + // best-effort cleanup + } + } + + const expectedKeywords = tc.expected_response_contains ?? []; + const lowerResponse = responseText.toLowerCase(); + const matchedKeywords = expectedKeywords.filter((kw) => lowerResponse.includes(kw.toLowerCase())); + const missedKeywords = expectedKeywords.filter((kw) => !lowerResponse.includes(kw.toLowerCase())); + + const passed = expectedKeywords.length === 0 || missedKeywords.length === 0; + + results.push({ + utterance: tc.utterance, + response: responseText, + passed, + matchedKeywords, + missedKeywords, + }); + } + + const passCount = results.filter((r) => r.passed).length; + const score = testCases.length > 0 ? passCount / testCases.length : 0; + return { score, passCount, results }; +} + +// --------------------------------------------------------------------------- +// Mutation proposal +// --------------------------------------------------------------------------- + +function stripCodeFences(text: string): string { + let stripped = text.trim(); + stripped = stripped.replace(/^```(?:yaml|agent|agentscript)?\s*\n?/i, ''); + stripped = stripped.replace(/\n?```\s*$/, ''); + return stripped.trim(); +} + +function buildMutationPrompt(agentContent: string, evalResults: TestCaseResult[]): string { + const failures = evalResults + .filter((r) => !r.passed) + .map( + (r) => + `- Utterance: "${r.utterance}"\n Response: "${r.response.slice( + 0, + 200 + )}"\n Missing keywords: ${r.missedKeywords.join(', ')}` + ) + .join('\n'); + + const successes = evalResults + .filter((r) => r.passed) + .map((r) => `- Utterance: "${r.utterance}" — PASSED`) + .join('\n'); + + return `You are an Agentforce agent optimization expert. You must improve the agent's instructions to fix failing test cases while preserving passing ones. + +CRITICAL RULES: +- Output ONLY the raw agent file content. No markdown fences, no explanations, no commentary. +- Preserve the EXACT file structure: system:, config:, variables:, language:, start_agent, topic blocks. +- Only modify text inside "instructions:" fields and "description:" fields. +- Do NOT add new topics, variables, or actions unless absolutely necessary. +- Do NOT change the config: block or default_agent_user. +- Keep all indentation and block-scalar markers (| and ->) exactly as they are. + +## Current Agent File (this is valid AgentScript — preserve its structure) +${agentContent} + +## Test Results +### Failing (needs improvement): +${failures || '(none)'} + +### Passing (preserve these): +${successes || '(none)'} + +## What to change +Improve ONLY the reasoning instructions text to better handle failing utterances. For example: +- Add explicit routing rules like "Route return/refund requests to the returns topic" +- Add keywords the agent should mention in responses (e.g., "Always mention refund policy when discussing returns") +- Make topic descriptions more specific so routing works correctly + +Output the complete agent file now:`; +} + +// --------------------------------------------------------------------------- +// Command +// --------------------------------------------------------------------------- + +export default class AgentOptimizeRun extends SfCommand { + public static readonly summary = messages.getMessage('summary'); + public static readonly description = messages.getMessage('description'); + public static readonly examples = messages.getMessages('examples'); + public static readonly requiresProject = true; + public static state = 'beta'; + + public static readonly envVariablesSection = toHelpSection( + 'ENVIRONMENT VARIABLES', + EnvironmentVariable.SF_TARGET_ORG + ); + + public static readonly flags = { + 'target-org': Flags.requiredOrg(), + 'api-version': Flags.orgApiVersion(), + 'authoring-bundle': Flags.string({ + summary: messages.getMessage('flags.authoring-bundle.summary'), + required: true, + }), + spec: Flags.file({ + char: 's', + required: true, + summary: messages.getMessage('flags.spec.summary'), + exists: true, + }), + iterations: Flags.integer({ + char: 'i', + default: 3, + summary: messages.getMessage('flags.iterations.summary'), + }), + }; + + public async run(): Promise { + const { flags } = await this.parse(AgentOptimizeRun); + const org = flags['target-org']; + const conn = org.getConnection(flags['api-version']); + const bundleName = flags['authoring-bundle']; + const maxIterations = flags.iterations; + + // 1. Load spec (supports JSON and YAML) + let spec: OptimizationSpec; + try { + const raw = await readFile(flags.spec, 'utf-8'); + if (flags.spec.endsWith('.yml') || flags.spec.endsWith('.yaml')) { + spec = parseYaml(raw) as OptimizationSpec; + } else { + spec = JSON.parse(raw) as OptimizationSpec; + } + } catch (e) { + throw new SfError(messages.getMessage('error.invalidSpec', [(e as Error).message])); + } + if (!spec.test_cases || !Array.isArray(spec.test_cases) || spec.test_cases.length === 0) { + throw new SfError(messages.getMessage('error.invalidSpec', ['missing or empty "test_cases" array'])); + } + + // 2. Locate agent file + const agentFile = locateAgentFile(this.project!, bundleName); + let agentContent: string; + try { + agentContent = await readFile(agentFile, 'utf-8'); + } catch { + throw new SfError(messages.getMessage('error.agentFileNotFound', [bundleName])); + } + const originalContent = agentContent; + + // 3. Baseline evaluation + this.log(`Evaluating baseline (${spec.test_cases.length} test cases)...`); + const baseline = await evaluateAgent(conn, this.project!, bundleName, spec.test_cases, (msg) => this.log(msg)); + this.log( + messages.getMessage('output.baseline', [ + `${(baseline.score * 100).toFixed(0)}%`, + baseline.passCount.toString(), + spec.test_cases.length.toString(), + ]) + ); + + // 4. Optimization loop + let bestScore = baseline.score; + let bestContent = agentContent; + let currentResults = baseline.results; + const iterations: IterationResult[] = []; + let keptCount = 0; + + for (let i = 1; i <= maxIterations; i++) { + this.log(`\nIteration ${i}/${maxIterations}...`); + + // 4a. If all tests pass, stop early + if (bestScore >= 1.0) { + this.log('All test cases passing — stopping early.'); + break; + } + + // 4b. Ask LLM for mutation + this.log(' Proposing improvements via LLM...'); + let mutatedContent: string; + try { + const prompt = buildMutationPrompt(agentContent, currentResults); + // eslint-disable-next-line no-await-in-loop + mutatedContent = await callLlm(conn, prompt); + mutatedContent = stripCodeFences(mutatedContent); + } catch (e) { + this.log(` LLM call failed: ${(e as Error).message}. Skipping iteration.`); + iterations.push({ + iteration: i, + mutation: 'LLM_ERROR', + score: bestScore, + passCount: baseline.passCount, + totalCount: spec.test_cases.length, + decision: 'REJECT', + }); + continue; + } + + // 4c. Basic validation: ensure it looks like an agent file + if (!mutatedContent.includes('system:') || !mutatedContent.includes('start_agent')) { + this.log(' LLM returned invalid agent content (missing system: or start_agent). Skipping iteration.'); + iterations.push({ + iteration: i, + mutation: 'INVALID_CONTENT', + score: bestScore, + passCount: baseline.passCount, + totalCount: spec.test_cases.length, + decision: 'REJECT', + }); + continue; + } + + // 4d. Apply mutation + // eslint-disable-next-line no-await-in-loop + await writeFile(agentFile, mutatedContent, 'utf-8'); + + // 4e. Evaluate + this.log(' Evaluating...'); + // eslint-disable-next-line no-await-in-loop + const evalResult = await evaluateAgent(conn, this.project!, bundleName, spec.test_cases, (msg) => this.log(msg)); + + // 4f. KEEP or REJECT + const mutationLabel = 'instruction_refine'; + if (evalResult.score > bestScore) { + bestScore = evalResult.score; + bestContent = mutatedContent; + agentContent = mutatedContent; + currentResults = evalResult.results; + keptCount++; + this.log( + messages.getMessage('output.iterationKeep', [ + i.toString(), + `${(evalResult.score * 100).toFixed(0)}%`, + `${(bestScore * 100).toFixed(0)}%`, + mutationLabel, + ]) + ); + iterations.push({ + iteration: i, + mutation: mutationLabel, + score: evalResult.score, + passCount: evalResult.passCount, + totalCount: spec.test_cases.length, + decision: 'KEEP', + }); + } else { + // Revert to best known version + // eslint-disable-next-line no-await-in-loop + await writeFile(agentFile, bestContent, 'utf-8'); + agentContent = bestContent; + this.log( + messages.getMessage('output.iterationReject', [ + i.toString(), + `${(evalResult.score * 100).toFixed(0)}%`, + `${(bestScore * 100).toFixed(0)}%`, + mutationLabel, + ]) + ); + iterations.push({ + iteration: i, + mutation: mutationLabel, + score: evalResult.score, + passCount: evalResult.passCount, + totalCount: spec.test_cases.length, + decision: 'REJECT', + }); + } + } + + // 5. Final write of best content + await writeFile(agentFile, bestContent, 'utf-8'); + + // 6. Summary + if (baseline.score >= 1.0 && iterations.length === 0) { + this.log(messages.getMessage('output.alreadyPerfect')); + } else if (bestScore > baseline.score) { + this.log( + `\n${messages.getMessage('output.summary', [ + `${(baseline.score * 100).toFixed(0)}%`, + `${(bestScore * 100).toFixed(0)}%`, + iterations.length.toString(), + keptCount.toString(), + ])}` + ); + this.log(messages.getMessage('output.agentUpdated', [agentFile])); + } else { + await writeFile(agentFile, originalContent, 'utf-8'); + this.log(`\n${messages.getMessage('output.noImprovement', [iterations.length.toString()])}`); + } + + return { + baselineScore: baseline.score, + bestScore, + iterations, + agentFile, + }; + } +} diff --git a/src/commands/agent/optimize/start.ts b/src/commands/agent/optimize/start.ts new file mode 100644 index 00000000..c4337c0a --- /dev/null +++ b/src/commands/agent/optimize/start.ts @@ -0,0 +1,167 @@ +/* + * Copyright 2026, Salesforce, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +import { readFile } from 'node:fs/promises'; +import { SfCommand, Flags } from '@salesforce/sf-plugins-core'; +import { Messages, SfError } from '@salesforce/core'; +import { AgentOptimization, type OptimizationExecution, type OptimizationStatus } from '@salesforce/agents'; + +Messages.importMessagesDirectoryFromMetaUrl(import.meta.url); +const messages = Messages.loadMessages('@salesforce/plugin-agent', 'agent.optimize.start'); + +export type AgentOptimizeStartResult = { + executionId: string; + status: string; + baselineScore?: number; + bestScore?: number; + iterationsRun?: number; +}; + +export default class AgentOptimizeStart extends SfCommand { + public static readonly summary = messages.getMessage('summary'); + public static readonly description = messages.getMessage('description'); + public static readonly examples = messages.getMessages('examples'); + public static readonly enableJsonFlag = true; + public static state = 'beta'; + + public static readonly flags = { + 'target-org': Flags.requiredOrg(), + 'api-version': Flags.orgApiVersion(), + 'authoring-bundle': Flags.string({ + summary: messages.getMessage('flags.authoring-bundle.summary'), + char: 'b', + required: true, + }), + criteria: Flags.file({ + summary: messages.getMessage('flags.criteria.summary'), + char: 'c', + required: true, + exists: true, + }), + 'test-cases': Flags.file({ + summary: messages.getMessage('flags.test-cases.summary'), + exists: true, + }), + 'max-iterations': Flags.integer({ + summary: messages.getMessage('flags.max-iterations.summary'), + default: 10, + }), + 'target-score': Flags.string({ + summary: messages.getMessage('flags.target-score.summary'), + default: '1.0', + }), + wait: Flags.integer({ + summary: messages.getMessage('flags.wait.summary'), + description: messages.getMessage('flags.wait.description'), + char: 'w', + }), + }; + + public async run(): Promise { + const { flags } = await this.parse(AgentOptimizeStart); + const connection = flags['target-org'].getConnection(flags['api-version']); + + let criteriaJson: string; + try { + criteriaJson = await readFile(flags.criteria, 'utf-8'); + } catch (e) { + throw new SfError(messages.getMessage('error.criteriaNotFound', [(e as Error).message])); + } + + let testCasesJson: string | undefined; + if (flags['test-cases']) { + try { + testCasesJson = await readFile(flags['test-cases'], 'utf-8'); + } catch (e) { + throw new SfError(messages.getMessage('error.invalidCriteria', [(e as Error).message])); + } + } + + let execution: OptimizationExecution; + try { + execution = await AgentOptimization.start(connection, { + authoringBundleName: flags['authoring-bundle'], + criteriaJson, + testCasesJson, + maxIterations: flags['max-iterations'], + targetScore: parseFloat(flags['target-score']), + }); + } catch (error) { + const wrapped = SfError.wrap(error); + throw new SfError(messages.getMessage('error.startFailed', [wrapped.message]), 'StartFailed', [], 4, wrapped); + } + + this.log(messages.getMessage('output.started', [execution.executionId])); + + if (!flags.wait) { + return { executionId: execution.executionId, status: 'NEW' }; + } + + const pollIntervalMs = 10_000; + const deadlineMs = Date.now() + flags.wait * 60_000; + + while (Date.now() < deadlineMs) { + // eslint-disable-next-line no-await-in-loop + await sleep(pollIntervalMs); + + let status: OptimizationStatus; + try { + // eslint-disable-next-line no-await-in-loop + status = await AgentOptimization.status(connection, execution.executionId); + } catch (error) { + const wrapped = SfError.wrap(error); + throw new SfError(messages.getMessage('error.pollFailed', [wrapped.message]), 'PollFailed', [], 4, wrapped); + } + + if (status.currentIteration != null && status.maxIterations != null) { + this.log( + messages.getMessage('output.progress', [ + String(status.currentIteration), + String(status.maxIterations), + status.baselineScore?.toFixed(2) ?? '-', + status.currentScore?.toFixed(2) ?? '-', + status.bestScore?.toFixed(2) ?? '-', + ]) + ); + } + + if (status.status === 'COMPLETED' || status.status === 'FAILED' || status.status === 'STOPPED_EARLY') { + this.log( + messages.getMessage('output.completed', [ + String(status.currentIteration ?? 0), + status.baselineScore?.toFixed(2) ?? '-', + status.bestScore?.toFixed(2) ?? '-', + ]) + ); + return { + executionId: execution.executionId, + status: status.status, + baselineScore: status.baselineScore, + bestScore: status.bestScore, + iterationsRun: status.currentIteration, + }; + } + } + + this.log(messages.getMessage('output.timeout', [String(flags.wait), execution.executionId])); + return { executionId: execution.executionId, status: 'TIMEOUT' }; + } +} + +function sleep(ms: number): Promise { + return new Promise((resolve) => { + setTimeout(resolve, ms); + }); +} diff --git a/src/commands/agent/optimize/status.ts b/src/commands/agent/optimize/status.ts new file mode 100644 index 00000000..5914e05e --- /dev/null +++ b/src/commands/agent/optimize/status.ts @@ -0,0 +1,71 @@ +/* + * Copyright 2026, Salesforce, Inc. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +import { SfCommand, Flags } from '@salesforce/sf-plugins-core'; +import { Messages, SfError } from '@salesforce/core'; +import { AgentOptimization, type OptimizationStatus } from '@salesforce/agents'; + +Messages.importMessagesDirectoryFromMetaUrl(import.meta.url); +const messages = Messages.loadMessages('@salesforce/plugin-agent', 'agent.optimize.status'); + +export type AgentOptimizeStatusResult = OptimizationStatus; + +export default class AgentOptimizeStatus extends SfCommand { + public static readonly summary = messages.getMessage('summary'); + public static readonly description = messages.getMessage('description'); + public static readonly examples = messages.getMessages('examples'); + public static readonly enableJsonFlag = true; + + public static readonly flags = { + 'target-org': Flags.requiredOrg(), + 'api-version': Flags.orgApiVersion(), + 'execution-id': Flags.string({ + summary: messages.getMessage('flags.execution-id.summary'), + char: 'i', + required: true, + }), + }; + + public async run(): Promise { + const { flags } = await this.parse(AgentOptimizeStatus); + const connection = flags['target-org'].getConnection(flags['api-version']); + + let result: OptimizationStatus; + try { + result = await AgentOptimization.status(connection, flags['execution-id']); + } catch (error) { + const wrapped = SfError.wrap(error); + throw new SfError(messages.getMessage('error.statusFailed', [wrapped.message]), 'StatusFailed', [], 4, wrapped); + } + + this.log(`Execution: ${result.executionId}`); + this.log(`Status: ${result.status}`); + if (result.currentIteration != null && result.maxIterations != null) { + this.log(`Progress: ${result.currentIteration}/${result.maxIterations} iterations`); + } + const parts: string[] = []; + if (result.baselineScore != null) parts.push(`Baseline: ${result.baselineScore.toFixed(2)}`); + if (result.currentScore != null) parts.push(`Current: ${result.currentScore.toFixed(2)}`); + if (result.bestScore != null) { + let best = `Best: ${result.bestScore.toFixed(2)}`; + if (result.bestIteration != null) best += ` (iteration ${result.bestIteration})`; + parts.push(best); + } + if (parts.length > 0) this.log(parts.join(' ')); + if (result.message) this.log(result.message); + + return result; + } +} diff --git a/test/mock-projects/agent-generate-template/specs/optimize-spec.json b/test/mock-projects/agent-generate-template/specs/optimize-spec.json new file mode 100644 index 00000000..62d8ac2e --- /dev/null +++ b/test/mock-projects/agent-generate-template/specs/optimize-spec.json @@ -0,0 +1,20 @@ +{ + "test_cases": [ + { + "utterance": "What are the resort's check-in hours?", + "expected_response_contains": ["check-in", "hours"] + }, + { + "utterance": "I need help with my employee schedule", + "expected_response_contains": ["schedule"] + }, + { + "utterance": "I have a complaint about my room", + "expected_response_contains": ["sorry", "help"] + }, + { + "utterance": "What is the weather like today?", + "expected_response_contains": ["weather"] + } + ] +} diff --git a/test/mock-projects/agent-generate-template/specs/optimize-spec.yml b/test/mock-projects/agent-generate-template/specs/optimize-spec.yml new file mode 100644 index 00000000..0c3fb0d8 --- /dev/null +++ b/test/mock-projects/agent-generate-template/specs/optimize-spec.yml @@ -0,0 +1,18 @@ +test_cases: + - utterance: "What are the resort's check-in hours?" + expected_response_contains: + - check-in + - hours + + - utterance: 'I need help with my employee schedule' + expected_response_contains: + - schedule + + - utterance: 'I have a complaint about my room' + expected_response_contains: + - sorry + - help + + - utterance: 'What is the weather like today?' + expected_response_contains: + - weather