From c834fb3953fcbf821e04af3051beada769e447b7 Mon Sep 17 00:00:00 2001 From: Brent Rager Date: Sun, 30 Aug 2026 17:01:43 -0400 Subject: [PATCH] th-ebe27d: let a host supply the turn's memory in the C#/Python/TS/Go servers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rust #330 put memory_for_access on StorageAdapter and had the server runner thread the result into the engine's agent options — the seam that lights up Big Smooth's durable auto-recall. The four sibling servers never did. The gap was invisible because BOTH ends were already built: all five engine cores implement Memory and already recall relevant entries into context. What was missing was the wire between them, so no matter what store a deployment had, every turn on these servers ran without auto-recall. Each server now takes a MemoryProvider (IMemoryProvider in C#) with one method, memory_for_access(access), resolved per turn and passed to the engine as AgentOptions.memory: C# IMemoryProvider DI (services.GetService()) Python MemoryProvider ServerState.memory_provider TypeScript MemoryProvider serve({ memoryProvider }) Go MemoryProvider WithMemoryProvider(...) access is threaded exactly as it is for knowledge, so a multi-tenant host can bind memory to the requester's org/user; single-tenant hosts (Big Smooth's daemon, the reason the seam exists) ignore it, so each language also ships a StaticMemoryProvider over one store. Nothing changes for anyone who does not opt in, and that is a test rather than a claim: each language asserts BOTH the no-provider and the declining-provider paths inject nothing, alongside the positive case, a relevance case (an unrelated message recalls nothing — not a blanket dump of every memory into every turn), and one proving the access argument actually reaches the provider. Five tests per language, named after their Rust counterparts in rust/smooth-operator-server/tests/injection_seams.rs. All four mutation-checked: dropping the single line that hands memory to the engine fails them. Go and Python attach the store post-construction (alongside hooks) rather than growing an 11- and 15-parameter constructor further — the pattern those files already use for exactly this reason. Verified: C# 412 passed; Python 389 passed / 21 skipped, ruff clean; TypeScript 43 files passed, tsc clean; Go full server suite green. Two notes for whoever picks this up next. The recall block's header text is deliberately NOT asserted — the five cores inject three different strings for it (th-ffaeae) — so the tests assert the recalled CONTENT reaches the model, which is the behavior the seam exists for. And the bundled lexical scorer counts raw token overlap with no stopword filter, so a single shared "the" scores a hit; worth knowing before trusting recall precision in production. No wire-protocol change. --- .changeset/memory-provider-seam.md | 46 +++++++ .../SmoothOperatorWebSocketExtensions.cs | 5 +- dotnet/server/src/FrameDispatcher.cs | 14 +- dotnet/server/src/Memory.cs | 42 ++++++ dotnet/server/src/TurnRunner.cs | 10 +- dotnet/server/tests/MemoryProviderTests.cs | 124 ++++++++++++++++++ go/server/dispatcher.go | 10 ++ go/server/memory.go | 39 ++++++ go/server/memory_test.go | 120 +++++++++++++++++ go/server/server.go | 9 ++ go/server/turn_runner.go | 8 ++ .../src/smooth_operator_server/__init__.py | 3 + .../src/smooth_operator_server/dispatcher.py | 9 ++ .../src/smooth_operator_server/memory.py | 44 +++++++ .../src/smooth_operator_server/server.py | 6 + .../src/smooth_operator_server/turn_runner.py | 10 ++ python/server/tests/test_memory_provider.py | 116 ++++++++++++++++ python/uv.lock | 2 +- typescript/server/src/frameDispatcher.ts | 9 ++ typescript/server/src/index.ts | 1 + typescript/server/src/memory.ts | 38 ++++++ typescript/server/src/server.ts | 8 ++ typescript/server/src/turnRunner.ts | 13 +- typescript/server/test/memoryProvider.test.ts | 100 ++++++++++++++ 24 files changed, 780 insertions(+), 6 deletions(-) create mode 100644 .changeset/memory-provider-seam.md create mode 100644 dotnet/server/src/Memory.cs create mode 100644 dotnet/server/tests/MemoryProviderTests.cs create mode 100644 go/server/memory.go create mode 100644 go/server/memory_test.go create mode 100644 python/server/src/smooth_operator_server/memory.py create mode 100644 python/server/tests/test_memory_provider.py create mode 100644 typescript/server/src/memory.ts create mode 100644 typescript/server/test/memoryProvider.test.ts diff --git a/.changeset/memory-provider-seam.md b/.changeset/memory-provider-seam.md new file mode 100644 index 00000000..54668f32 --- /dev/null +++ b/.changeset/memory-provider-seam.md @@ -0,0 +1,46 @@ +--- +"@smooai/smooth-operator": patch +--- + +feat(dotnet,python,ts,go): let a host supply the turn's memory — durable auto-recall off Rust (th-ebe27d) + +Rust #330 put `memory_for_access` on `StorageAdapter` and had the server runner thread the +result into the engine's agent options, which is what lights up Big Smooth's durable +auto-recall. The four sibling servers never did — and the gap was invisible, because **all +five engine cores already implement `Memory` and already recall relevant entries into +context**. The capability was fully built on both ends with nothing connecting them: no +matter what store a deployment had, every turn on these servers ran without auto-recall. + +Each server now takes a `MemoryProvider` (`IMemoryProvider` in C#) with one method — +`memory_for_access(access)` — resolved per turn and passed to the engine as +`AgentOptions.memory`: + +| | seam | install | +|---|---|---| +| C# | `IMemoryProvider` | DI (`services.GetService()`) | +| Python | `MemoryProvider` | `ServerState.memory_provider` | +| TypeScript | `MemoryProvider` | `serve({ memoryProvider })` | +| Go | `MemoryProvider` | `WithMemoryProvider(...)` | + +`access` is threaded exactly as it is for knowledge, so a multi-tenant host can bind memory +to the requester's org/user; single-tenant hosts — Big Smooth's daemon, the reason the seam +exists — ignore it, so each language also ships a `StaticMemoryProvider` over one store. + +**Nothing changes for anyone who does not opt in.** No provider, or a provider that returns +nothing for this caller, leaves the turn byte-for-byte what it was — and that is a test, not +a claim: each language asserts the no-provider and the declining-provider paths inject +nothing, alongside the positive case and a relevance case (an unrelated message recalls +nothing, so this is not a blanket dump of every stored memory into every turn). + +Five tests per language, named after their Rust counterparts in +`rust/smooth-operator-server/tests/injection_seams.rs`, all four mutation-checked — dropping +the one line that hands memory to the engine fails them. + +Two notes for whoever picks this up next. The recall block's **header text is deliberately +not asserted**: the five cores currently inject three different strings for it (th-ffaeae), +so the tests assert the recalled *content* reaches the model, which is the behavior the seam +exists for. And the bundled lexical scorer counts raw token overlap with **no stopword +filter**, so in practice a single shared "the" scores a hit — worth knowing before trusting +recall precision in production. + +No wire-protocol change. diff --git a/dotnet/server/aspnetcore/SmoothOperatorWebSocketExtensions.cs b/dotnet/server/aspnetcore/SmoothOperatorWebSocketExtensions.cs index 1164efb9..5af7de19 100644 --- a/dotnet/server/aspnetcore/SmoothOperatorWebSocketExtensions.cs +++ b/dotnet/server/aspnetcore/SmoothOperatorWebSocketExtensions.cs @@ -123,7 +123,10 @@ private static FrameDispatcher BuildDispatcher(HttpContext context) // fall back to the env-configured directory resolver (SMOOTH_SKILLS_DIR). Unset ⇒ null ⇒ any // `skill` field is a clean SKILL_NOT_FOUND, so a multi-tenant deploy never serves host skills // by accident. Mirrors Rust's install_skill_resolver_from_env. - skillResolver: services.GetService() ?? DirSkillResolver.FromEnv()); + skillResolver: services.GetService() ?? DirSkillResolver.FromEnv(), + // Durable auto-recall (th-ebe27d / Rust #330). A host registers an IMemoryProvider — or a + // StaticMemoryProvider over one store — to light it up; unregistered ⇒ no auto-recall. + memoryProvider: services.GetService()); } private static async Task PumpAsync( diff --git a/dotnet/server/src/FrameDispatcher.cs b/dotnet/server/src/FrameDispatcher.cs index b16b73af..5a1b8905 100644 --- a/dotnet/server/src/FrameDispatcher.cs +++ b/dotnet/server/src/FrameDispatcher.cs @@ -59,6 +59,9 @@ public sealed class FrameDispatcher private readonly TurnLimits _limits; private readonly ILogger? _logger; private readonly ISkillResolver? _skillResolver; + /// Supplies the turn's durable-recall store, scoped to this connection's access + /// (th-ebe27d / Rust #330). Null → no auto-recall, the default. + private readonly IMemoryProvider? _memoryProvider; // The connection's SINGLE in-flight send_message turn, if one is running. A turn that calls a // confirmation-gated tool parks awaiting a later confirm_tool_action frame, so the turn runs as a @@ -102,7 +105,8 @@ public FrameDispatcher( TurnLimits? limits = null, ILogger? logger = null, IReadOnlyList? toolHooks = null, - ISkillResolver? skillResolver = null) + ISkillResolver? skillResolver = null, + IMemoryProvider? memoryProvider = null) { _store = store ?? throw new ArgumentNullException(nameof(store)); _chatClient = chatClient ?? throw new ArgumentNullException(nameof(chatClient)); @@ -110,6 +114,7 @@ public FrameDispatcher( _access = access ?? AccessContext.Anonymous; _systemPrompt = systemPrompt; _skillResolver = skillResolver; + _memoryProvider = memoryProvider; _reranker = reranker; _tools = tools ?? Array.Empty(); // Tool-call hooks (surveillance / redaction) forwarded to every turn's registry (empty → no @@ -784,6 +789,11 @@ private async Task HandleSendMessageAsync(JsonObject frame, string? requestId, A // through the same ACL-filtered store (a doc the caller's groups don't grant is never a candidate). var scopedKnowledge = _knowledge?.ForAccess(_access); + // Durable auto-recall, scoped the same way retrieval is: the host decides which memory this + // caller recalls from. Null (no provider installed, or one that declines this access) leaves + // the turn without auto-recall — byte-for-byte unchanged. + var scopedMemory = _memoryProvider?.MemoryForAccess(_access); + // Built-in knowledge_search: a model-callable search over the connection's ACL-scoped knowledge // (parity with the Rust server's KnowledgeSearchTool). Prepended before the enabled_tools filter // so it flows through the SAME per-agent restriction + auth gate as every other tool — an agent @@ -824,7 +834,7 @@ private async Task HandleSendMessageAsync(JsonObject frame, string? requestId, A // have arrived on a PREVIOUS connection (a reconnect resumes the conversation on a fresh // dispatcher), and a per-connection map would read empty there. th-13df6d. var capabilities = await _store.GetClientSupportsAsync(session.ConversationId, cancellationToken).ConfigureAwait(false); - var runner = new TurnRunner(_chatClient, _store, scopedKnowledge, _systemPrompt, _reranker, gatedTools, confirmTools, _confirmations, agentConfig, _judge, _limits, _logger, toolHooks: _toolHooks, interactions: _interactions, interactionPark: _interactionPark, capabilities: capabilities, interactionEffects: _sessionIdentity) + var runner = new TurnRunner(_chatClient, _store, scopedKnowledge, _systemPrompt, _reranker, gatedTools, confirmTools, _confirmations, agentConfig, _judge, _limits, _logger, toolHooks: _toolHooks, interactions: _interactions, interactionPark: _interactionPark, capabilities: capabilities, interactionEffects: _sessionIdentity, memory: scopedMemory) { ConfirmationTimeout = ConfirmationTimeout, }; diff --git a/dotnet/server/src/Memory.cs b/dotnet/server/src/Memory.cs new file mode 100644 index 00000000..bb1c5fd5 --- /dev/null +++ b/dotnet/server/src/Memory.cs @@ -0,0 +1,42 @@ +using SmooAI.SmoothOperator.Core; + +namespace SmooAI.SmoothOperator.Server; + +/// +/// Supplies the durable-recall handle for a turn — the C# analog of the Rust +/// StorageAdapter::memory_for_access seam (PR #330). +/// +/// The engine already knows how to auto-recall: give AgentOptions.Memory a store and it +/// pulls the entries relevant to the user's message into the turn's context. What was missing on +/// this server is the way for a HOST to say which store — so every turn ran without +/// auto-recall regardless of what the deployment had. +/// +/// The access argument is threaded (mirroring IKnowledgeBase.ForAccess) so a +/// multi-tenant backend can bind memory to the requester's org/user; a single-tenant host — Big +/// Smooth's daemon, which is the reason this seam exists — ignores it and returns its one store. +/// +public interface IMemoryProvider +{ + /// + /// The memory to auto-recall from for a caller with this access, or null for none. + /// null is the default for every deployment that has not opted in, and leaves the turn + /// byte-for-byte unchanged. + /// + IAgentMemory? MemoryForAccess(AccessContext access); +} + +/// +/// An over one unscoped store — the single-tenant case (Big Smooth's +/// daemon hands its SQLite-backed store straight through). A multi-tenant host implements the +/// interface itself and keys off access instead. +/// +public sealed class StaticMemoryProvider : IMemoryProvider +{ + private readonly IAgentMemory? _memory; + + /// The store every caller recalls from; null disables auto-recall. + public StaticMemoryProvider(IAgentMemory? memory) => _memory = memory; + + /// + public IAgentMemory? MemoryForAccess(AccessContext access) => _memory; +} diff --git a/dotnet/server/src/TurnRunner.cs b/dotnet/server/src/TurnRunner.cs index 0724ba76..87a2165e 100644 --- a/dotnet/server/src/TurnRunner.cs +++ b/dotnet/server/src/TurnRunner.cs @@ -62,6 +62,9 @@ public sealed class TurnRunner private readonly IChatClient _chatClient; private readonly ISessionStore _store; private readonly IKnowledgeBase? _knowledge; + /// Durable-recall store for this turn, already resolved for the connection's access by + /// the dispatcher (th-ebe27d / Rust #330). Null → the turn runs without auto-recall. + private readonly IAgentMemory? _memory; private readonly IReranker? _reranker; private readonly string _systemPrompt; private readonly IReadOnlyList _tools; @@ -96,11 +99,12 @@ public sealed class TurnRunner /// — or a test — narrows it). public TimeSpan ConfirmationTimeout { get; init; } = DefaultConfirmationTimeout; - public TurnRunner(IChatClient chatClient, ISessionStore store, IKnowledgeBase? knowledge = null, string? systemPrompt = null, IReranker? reranker = null, IReadOnlyList? tools = null, IReadOnlyList? confirmTools = null, ConfirmationRegistry? confirmations = null, AgentConfig? agentConfig = null, IWorkflowJudge? judge = null, TurnLimits? limits = null, ILogger? logger = null, IChatClient? preambleChatClient = null, IReadOnlyList? toolHooks = null, InteractionCatalog? interactions = null, InteractionParkRegistry? interactionPark = null, IReadOnlyCollection? capabilities = null, SessionIdentityRegistry? interactionEffects = null) + public TurnRunner(IChatClient chatClient, ISessionStore store, IKnowledgeBase? knowledge = null, string? systemPrompt = null, IReranker? reranker = null, IReadOnlyList? tools = null, IReadOnlyList? confirmTools = null, ConfirmationRegistry? confirmations = null, AgentConfig? agentConfig = null, IWorkflowJudge? judge = null, TurnLimits? limits = null, ILogger? logger = null, IChatClient? preambleChatClient = null, IReadOnlyList? toolHooks = null, InteractionCatalog? interactions = null, InteractionParkRegistry? interactionPark = null, IReadOnlyCollection? capabilities = null, SessionIdentityRegistry? interactionEffects = null, IAgentMemory? memory = null) { _chatClient = chatClient ?? throw new ArgumentNullException(nameof(chatClient)); _store = store ?? throw new ArgumentNullException(nameof(store)); _knowledge = knowledge; + _memory = memory; _reranker = reranker; _systemPrompt = systemPrompt ?? "You are a helpful customer support agent. Answer using only the knowledge provided to you; if it is not there, say you don't know."; @@ -331,6 +335,10 @@ public async Task RunAsync(string conversationId, string requestId, MaxIterations = _limits.MaxIterations, MaxOutputTokens = _limits.MaxTokens, ModelMaxOutputTokens = _limits.ModelMaxOutputTokens, + // Durable auto-recall: with a store attached the engine pulls the entries relevant to the + // user's message into context. Null (every deployment that has not opted in) leaves the + // turn byte-for-byte unchanged. + Memory = _memory, }; foreach (var tool in _tools) { diff --git a/dotnet/server/tests/MemoryProviderTests.cs b/dotnet/server/tests/MemoryProviderTests.cs new file mode 100644 index 00000000..bb7ab73e --- /dev/null +++ b/dotnet/server/tests/MemoryProviderTests.cs @@ -0,0 +1,124 @@ +using System.Text.Json.Nodes; +using SmooAI.SmoothOperator.Core; +using SmooAI.SmoothOperator.Server; + +namespace SmooAI.SmoothOperator.Server.Tests; + +/// +/// Durable auto-recall parity with the Rust reference (PR #330 — the +/// StorageAdapter::memory_for_access seam, tested in +/// rust/smooth-operator-server/tests/injection_seams.rs). +/// +/// The engine already knew how to recall; what was missing on this server was the host's way to say +/// WHICH store, so every turn ran without auto-recall no matter what the deployment had. These tests +/// are named after their Rust counterparts so a parity gap stays visible. +/// +/// The recall block's header text is deliberately NOT asserted here: the five cores currently inject +/// three different strings for it (th-ffaeae). The assertion is on the recalled CONTENT reaching the +/// model, which is the behavior the seam exists for. +/// +public class MemoryProviderTests +{ + private static async Task CreateSessionAsync(FrameDispatcher dispatcher, List events) + { + await dispatcher.DispatchAsync("""{"action":"create_conversation_session","agentId":"11111111-1111-1111-1111-111111111111","requestId":"r1"}""", events.Add); + var sessionId = events[0]["data"]!["sessionId"]!.GetValue(); + events.Clear(); + return sessionId; + } + + /// Everything the model was sent this turn, flattened — the surface a recalled memory + /// must show up in. + private static string AllContentSeen(RecordingChatClient chat) => + string.Join("\n", chat.LastMessages.Select(m => m.Text)); + + private static async Task RunTurnAsync(IMemoryProvider? provider, string message) + { + var chat = new RecordingChatClient("ok"); + var dispatcher = new FrameDispatcher(new InMemorySessionStore(), chat, memoryProvider: provider); + var events = new List(); + var sessionId = await CreateSessionAsync(dispatcher, events); + + await dispatcher.DispatchAsync( + $$"""{"action":"send_message","requestId":"r2","sessionId":"{{sessionId}}","message":"{{message}}"}""", + events.Add); + await dispatcher.WaitForTurnsAsync(); + return chat; + } + + // ── rust: no_memory_means_no_recall_injection ──────────────────────────── + + /// Default: no provider ⇒ no auto-recall. Guards against the seam injecting when absent — + /// an unopted deployment's turn must be byte-for-byte what it was before. + [Fact] + public async Task NoMemoryMeansNoRecallInjection() + { + var chat = await RunTurnAsync(null, "add shows to my watchlist"); + Assert.DoesNotContain("smoo-hub watchlist", AllContentSeen(chat), StringComparison.Ordinal); + } + + /// A provider that returns null for this caller is the same as no provider — the seam + /// must not fabricate a store just because one was installed. + [Fact] + public async Task ProviderReturningNullMeansNoRecallInjection() + { + var chat = await RunTurnAsync(new StaticMemoryProvider(null), "add shows to my watchlist"); + Assert.DoesNotContain("smoo-hub watchlist", AllContentSeen(chat), StringComparison.Ordinal); + } + + // ── rust: attached_memory_is_auto_recalled_into_the_turn ───────────────── + + /// With a store attached the engine recalls the entries relevant to the user's message + /// and injects them into the turn — the seam that lights up Big Smooth's durable auto-recall. + [Fact] + public async Task AttachedMemoryIsAutoRecalledIntoTheTurn() + { + var memory = new InMemoryAgentMemory(); + await memory.StoreAsync(new MemoryEntry("m-1", "always add shows to the smoo-hub watchlist", MemoryType.Project)); + + // The message shares "add", "shows", "watchlist" with the stored entry, so the engine's + // word-overlap recall surfaces it. + var chat = await RunTurnAsync(new StaticMemoryProvider(memory), "add shows to my watchlist"); + + Assert.Contains("smoo-hub watchlist", AllContentSeen(chat), StringComparison.Ordinal); + } + + /// An unrelated message recalls nothing: the seam is relevance-gated by the engine, not a + /// blanket dump of every stored memory into every turn. The message shares NO token with the entry — + /// the bundled lexical scorer counts raw token overlap with no stopword filter, so a single shared + /// "the" is enough to score a hit. + [Fact] + public async Task IrrelevantMessageRecallsNothing() + { + var memory = new InMemoryAgentMemory(); + await memory.StoreAsync(new MemoryEntry("m-1", "always add shows to the smoo-hub watchlist", MemoryType.Project)); + + var chat = await RunTurnAsync(new StaticMemoryProvider(memory), "explain quantum entanglement"); + + Assert.DoesNotContain("smoo-hub watchlist", AllContentSeen(chat), StringComparison.Ordinal); + } + + /// The seam is access-scoped (mirroring IKnowledgeBase.ForAccess) so a multi-tenant + /// host can bind memory to the requester — the argument must actually reach the provider. + [Fact] + public async Task ProviderSeesTheCallersAccess() + { + var seen = new List(); + var chat = await RunTurnAsync(new RecordingMemoryProvider(seen), "hello"); + + Assert.Single(seen); + } + + private sealed class RecordingMemoryProvider : IMemoryProvider + { + private readonly List _seen; + + public RecordingMemoryProvider(List seen) => _seen = seen; + + public IAgentMemory? MemoryForAccess(AccessContext access) + { + _seen.Add(access); + return null; + } + } +} diff --git a/go/server/dispatcher.go b/go/server/dispatcher.go index 6182d526..7785744f 100644 --- a/go/server/dispatcher.go +++ b/go/server/dispatcher.go @@ -39,6 +39,10 @@ type FrameDispatcher struct { // is long enough. Nil → the feature is off and any skill field is a clean // SKILL_NOT_FOUND, so a multi-tenant deploy never serves host skills by accident. skills SkillResolver + // memoryProvider supplies the turn's durable-recall store, scoped to this connection's + // access (th-ebe27d / Rust #330). Set by the server after construction, alongside skills. + // nil → no auto-recall, the default. + memoryProvider MemoryProvider // associate records this connection's backplane targets as they are learned (set by // the connection loop). Nil → session/agent targets are never routable. associate func(target Target) @@ -790,6 +794,12 @@ func (d *FrameDispatcher) handleSendMessage(ctx context.Context, frame inboundFr }() runner := NewTurnRunner(d.client, d.store, effectiveSystemPrompt, d.knowledge, effectiveTools, d.confirmTools, d.confirmations, workflow, session.CurrentStepID, d.judgeModel, d.modelCeiling) runner.hooks = d.hooks + // Durable auto-recall, scoped the way retrieval is: the host decides which memory this + // caller recalls from. nil (no provider, or one that declines this access) leaves the turn + // without auto-recall — byte-for-byte unchanged. + if d.memoryProvider != nil { + runner.memory = d.memoryProvider.MemoryForAccess(d.access) + } // Rich Interactions: give the runner the hosted kinds, the park/resume registry, // and this session's declared capabilities, so it registers one raise tool per // kind (rich park when the capability is declared, else conversational fallback). diff --git a/go/server/memory.go b/go/server/memory.go new file mode 100644 index 00000000..78d9efb2 --- /dev/null +++ b/go/server/memory.go @@ -0,0 +1,39 @@ +package server + +import ( + core "github.com/SmooAI/smooth-operator-core/go/core" +) + +// Durable auto-recall — the server side of the engine's Memory seam (Rust PR #330). +// +// The engine already knows how to auto-recall: give AgentOptions.Memory a store and it pulls the +// entries relevant to the user's message into the turn's context. What was missing on this server +// is the way for a HOST to say WHICH store — so every turn ran without auto-recall regardless of +// what the deployment had. +// +// Mirrors the Rust StorageAdapter::memory_for_access seam. access is threaded (as it is for +// knowledge) so a multi-tenant backend can bind memory to the requester's org/user; a single-tenant +// host — Big Smooth's daemon, the reason this seam exists — ignores it and returns its one store. + +// MemoryProvider supplies the durable-recall handle for a turn. +type MemoryProvider interface { + // MemoryForAccess returns the memory to auto-recall from for a caller with this access, or nil + // for none. nil is the default for every deployment that has not opted in, and leaves the turn + // byte-for-byte unchanged. + MemoryForAccess(access AccessContext) core.Memory +} + +// StaticMemoryProvider is a MemoryProvider over one unscoped store — the single-tenant case (Big +// Smooth's daemon hands its SQLite-backed store straight through). A multi-tenant host implements +// the interface itself and keys off access instead. +type StaticMemoryProvider struct { + memory core.Memory +} + +// NewStaticMemoryProvider wraps one store (nil disables auto-recall). +func NewStaticMemoryProvider(memory core.Memory) *StaticMemoryProvider { + return &StaticMemoryProvider{memory: memory} +} + +// MemoryForAccess implements MemoryProvider. +func (p *StaticMemoryProvider) MemoryForAccess(_ AccessContext) core.Memory { return p.memory } diff --git a/go/server/memory_test.go b/go/server/memory_test.go new file mode 100644 index 00000000..1dd5f305 --- /dev/null +++ b/go/server/memory_test.go @@ -0,0 +1,120 @@ +package server + +import ( + "encoding/json" + "strings" + "testing" + + core "github.com/SmooAI/smooth-operator-core/go/core" +) + +// Durable auto-recall parity (th-ebe27d / Rust PR #330 — the StorageAdapter::memory_for_access +// seam, tested in rust/smooth-operator-server/tests/injection_seams.rs). +// +// The engine already knew how to recall; what was missing on this server was the host's way to say +// WHICH store, so every turn ran without auto-recall no matter what the deployment had. Tests are +// named after their Rust counterparts so a parity gap stays visible. +// +// The recall block's header text is deliberately NOT asserted: the five cores currently inject three +// different strings for it (th-ffaeae). The assertion is on the recalled CONTENT reaching the model, +// which is the behavior the seam exists for. + +const recalledFact = "always add shows to the smoo-hub watchlist" + +// allContentSeen flattens everything the model was sent this turn — the surface a recalled memory +// must show up in. +func allContentSeen(t *testing.T, mock *core.MockLlmProvider) string { + t.Helper() + var sb strings.Builder + for _, call := range mock.Calls() { + blob, err := json.Marshal(call.Messages) + if err != nil { + t.Fatalf("marshal messages: %v", err) + } + sb.Write(blob) + } + return sb.String() +} + +func memoryWithEntry() core.Memory { + m := &core.InMemoryMemory{} + m.Remember(recalledFact) + return m +} + +// runMemoryTurn drives one plain text turn with the given provider installed the way the server +// installs it (post-construction, alongside hooks). +func runMemoryTurn(t *testing.T, provider MemoryProvider, message string) *core.MockLlmProvider { + t.Helper() + mock := core.NewMockLlmProvider().PushText("ok") + d := NewFrameDispatcher(NewInMemorySessionStore(), mock, AccessContext{}, "BASE PROMPT", nil, nil, nil, nil, nil, "", nil, nil, nil, nil) + d.memoryProvider = provider + sid := createSessionForTest(t, d) + + sink := sendAndWait(d, `{"action":"send_message","requestId":"r-1","sessionId":"`+sid+`","message":"`+message+`"}`) + if sink.find("eventual_response") == nil { + t.Fatal("turn did not complete") + } + return mock +} + +// ── rust: no_memory_means_no_recall_injection ──────────────────────────────── + +// Default: no provider ⇒ no auto-recall. Guards against the seam injecting when absent — an unopted +// deployment's turn must be byte-for-byte what it was before. +func TestNoMemoryMeansNoRecallInjection(t *testing.T) { + mock := runMemoryTurn(t, nil, "add shows to my watchlist") + if strings.Contains(allContentSeen(t, mock), recalledFact) { + t.Error("a turn with no memory provider carried a recalled memory") + } +} + +// A provider returning nil is the same as no provider — the seam must not fabricate a store just +// because one was installed. +func TestProviderReturningNilMeansNoRecallInjection(t *testing.T) { + mock := runMemoryTurn(t, NewStaticMemoryProvider(nil), "add shows to my watchlist") + if strings.Contains(allContentSeen(t, mock), recalledFact) { + t.Error("a declining provider still injected a recalled memory") + } +} + +// ── rust: attached_memory_is_auto_recalled_into_the_turn ───────────────────── + +// With a store attached the engine recalls the entries relevant to the user's message and injects +// them into the turn — the seam that lights up Big Smooth's durable auto-recall. +func TestAttachedMemoryIsAutoRecalledIntoTheTurn(t *testing.T) { + // The message shares "add", "shows", "watchlist" with the stored entry, so the engine's + // word-overlap recall surfaces it. + mock := runMemoryTurn(t, NewStaticMemoryProvider(memoryWithEntry()), "add shows to my watchlist") + if !strings.Contains(allContentSeen(t, mock), recalledFact) { + t.Errorf("an attached memory was not recalled into the turn: %s", allContentSeen(t, mock)) + } +} + +// An unrelated message recalls nothing: relevance-gated by the engine, not a blanket dump of every +// stored memory into every turn. The message shares NO token with the entry — the bundled lexical +// scorer counts raw token overlap with no stopword filter, so a single shared "the" would be enough +// to score a hit. +func TestIrrelevantMessageRecallsNothing(t *testing.T) { + mock := runMemoryTurn(t, NewStaticMemoryProvider(memoryWithEntry()), "explain quantum entanglement") + if strings.Contains(allContentSeen(t, mock), recalledFact) { + t.Error("an unrelated message recalled a memory it shares no token with") + } +} + +// The seam is access-scoped (mirroring knowledge) so a multi-tenant host can bind memory to the +// requester — the argument must actually reach the provider. +func TestProviderSeesTheCallersAccess(t *testing.T) { + rec := &recordingMemoryProvider{} + runMemoryTurn(t, rec, "hello") + if rec.calls != 1 { + t.Fatalf("provider called %d times, want exactly 1", rec.calls) + } +} + +type recordingMemoryProvider struct{ calls int } + +func (r *recordingMemoryProvider) MemoryForAccess(_ AccessContext) core.Memory { + r.calls++ + return nil +} diff --git a/go/server/server.go b/go/server/server.go index 86cdc229..2e58472d 100644 --- a/go/server/server.go +++ b/go/server/server.go @@ -32,6 +32,10 @@ type Server struct { // Nil → fall back to the env-configured DirSkillResolver (SMOOTH_SKILLS_DIR); with // neither, any skill field is a clean SKILL_NOT_FOUND. skills SkillResolver + // memory supplies each turn's durable-recall store, scoped to the connection's access + // (th-ebe27d / Rust #330). Nil → no auto-recall, which is every deployment that has not + // opted in. + memory MemoryProvider // tools are registered with the agent on every turn (default none → no behavior // change). The dispatcher threads them into the turn runner, which passes them // straight to the engine AgentOptions; the engine drives the tool loop and the @@ -127,6 +131,10 @@ func WithSystemPrompt(p string) Option { return func(srv *Server) { srv.systemP // and any skill field is a clean SKILL_NOT_FOUND. func WithSkillResolver(r SkillResolver) Option { return func(srv *Server) { srv.skills = r } } +// WithMemoryProvider installs the provider that supplies each turn's durable-recall store. Omitted +// → no auto-recall, which is every deployment that has not opted in. +func WithMemoryProvider(p MemoryProvider) Option { return func(srv *Server) { srv.memory = p } } + // WithTools registers the engine tools the agent may call during a turn (default none). // Threaded into every turn via the dispatcher → turn runner → engine AgentOptions. func WithTools(tools []core.Tool) Option { return func(srv *Server) { srv.tools = tools } } @@ -360,6 +368,7 @@ func (s *Server) connectionLoop(conn *websocket.Conn, access AccessContext) { // construction (before the dispatcher serves any frame) to avoid churning the // long constructor signature. Nil → no hooks. dispatcher.hooks = s.hooks + dispatcher.memoryProvider = s.memory // Same post-construction seam as hooks. Explicit resolver wins, else the // env-configured default, else off — mirrors the Rust/TS rule. if s.skills != nil { diff --git a/go/server/turn_runner.go b/go/server/turn_runner.go index fc9b0dc0..05b818c7 100644 --- a/go/server/turn_runner.go +++ b/go/server/turn_runner.go @@ -77,6 +77,10 @@ type TurnRunner struct { // around every tool the agent dispatches this turn (nil → none). Set by the // dispatcher after construction, alongside tools. hooks []core.ToolHook + // memory is the durable-recall store for this turn, already resolved for the connection's + // access by the dispatcher (th-ebe27d / Rust #330). Set after construction alongside hooks — + // the constructor signature is long enough. nil → the turn runs without auto-recall. + memory core.Memory // knowledge is the retriever (already SCOPED to the connection's access) the agent // grounds on. When set, the runner also queries it with the user message (top // autoContextLimit) to build the turn's auto-context citations — the sources the @@ -248,6 +252,10 @@ func (r *TurnRunner) Run(ctx context.Context, sessionID, conversationID, request // citations built above. r.systemPrompt is already assembled (base + per-agent // config + current workflow step) by the caller. opts := core.AgentOptions{Instructions: r.systemPrompt, Tools: r.tools, Knowledge: r.knowledge, Hooks: r.hooks} + // Durable auto-recall: with a store attached the engine pulls the entries relevant to the + // user's message into context. nil (every deployment that has not opted in) leaves the turn + // byte-for-byte unchanged. + opts.Memory = r.memory // Raised, no-longer-starving turn defaults (8192 / 20), with max_tokens clamped to // the model's output ceiling when known (nil ⇒ the raised default). Setting these // explicitly (rather than relying on the engine defaults) keeps the server robust diff --git a/python/server/src/smooth_operator_server/__init__.py b/python/server/src/smooth_operator_server/__init__.py index 6d798ae5..c805ee50 100644 --- a/python/server/src/smooth_operator_server/__init__.py +++ b/python/server/src/smooth_operator_server/__init__.py @@ -37,6 +37,7 @@ InteractionRequest, PendingInteractions, ) +from .memory import MemoryProvider, StaticMemoryProvider from .otp import ( OtpChannel, OtpContact, @@ -86,6 +87,8 @@ "coding_tools_from_env", "resolve_workspace_path", "FrameDispatcher", + "MemoryProvider", + "StaticMemoryProvider", "SKILLS_DIR_ENV", "DirSkillResolver", "SkillResolver", diff --git a/python/server/src/smooth_operator_server/dispatcher.py b/python/server/src/smooth_operator_server/dispatcher.py index b9990c08..c8ecec48 100644 --- a/python/server/src/smooth_operator_server/dispatcher.py +++ b/python/server/src/smooth_operator_server/dispatcher.py @@ -33,6 +33,7 @@ from .backplane import Target from .confirmation import ConfirmationRegistry from .interaction import InteractionOutcome, InteractionRegistry, PendingInteractions +from .memory import MemoryProvider from .otp import OtpContact, OtpInvalid, OtpService, OtpVerified from .session_store import SessionStore from .skills import SkillResolver, resolve_section @@ -66,6 +67,7 @@ def __init__( otp_service: OtpService | None = None, associate: Callable[[Target], Awaitable[None]] | None = None, skill_resolver: SkillResolver | None = None, + memory_provider: MemoryProvider | None = None, ) -> None: self._store = store self._chat_client = chat_client @@ -76,6 +78,9 @@ def __init__( #: None → the feature is off and any `skill` field is a clean SKILL_NOT_FOUND, #: so a multi-tenant deploy never serves host skills by accident. self._skill_resolver = skill_resolver + #: Supplies the turn's durable-recall store, scoped to this connection's access + #: (th-ebe27d / Rust #330). None → no auto-recall, the default. + self._memory_provider = memory_provider self._model = model self._tools = tools or [] #: Tool-name patterns gated behind human confirmation (empty → HITL off). @@ -586,6 +591,10 @@ async def _handle_send_message(self, frame: dict, request_id: str | None, sink: self._chat_client, self._store, knowledge=self._knowledge, + # Durable auto-recall, scoped the way retrieval is: the host decides which memory + # this caller recalls from. None (no provider, or one that declines this access) + # leaves the turn without auto-recall — byte-for-byte unchanged. + memory=(self._memory_provider.memory_for_access(self._access) if self._memory_provider else None), system_prompt=self._system_prompt, skill_section=skill_section_for_turn, model=self._model, diff --git a/python/server/src/smooth_operator_server/memory.py b/python/server/src/smooth_operator_server/memory.py new file mode 100644 index 00000000..872d96bd --- /dev/null +++ b/python/server/src/smooth_operator_server/memory.py @@ -0,0 +1,44 @@ +"""Durable auto-recall — the server side of the engine's ``Memory`` seam (Rust PR #330). + +The engine already knows how to auto-recall: give ``AgentOptions.memory`` a store and it +pulls the entries relevant to the user's message into the turn's context. What was missing +on this server is the way for a HOST to say *which* store — so every turn ran without +auto-recall regardless of what the deployment had. + +Mirrors the Rust ``StorageAdapter::memory_for_access`` seam. ``access`` is threaded (as it is +for knowledge) so a multi-tenant backend can bind memory to the requester's org/user; a +single-tenant host — Big Smooth's daemon, the reason this seam exists — ignores it and +returns its one store. +""" + +from __future__ import annotations + +from typing import Protocol, runtime_checkable + +from smooth_operator_core import Memory + +from .auth import AccessContext + + +@runtime_checkable +class MemoryProvider(Protocol): + """Supplies the durable-recall handle for a turn.""" + + def memory_for_access(self, access: AccessContext) -> Memory | None: + """The memory to auto-recall from for a caller with this access, or ``None`` for none. + + ``None`` is the default for every deployment that has not opted in, and leaves the + turn byte-for-byte unchanged.""" + ... + + +class StaticMemoryProvider: + """A :class:`MemoryProvider` over one unscoped store — the single-tenant case (Big Smooth's + daemon hands its SQLite-backed store straight through). A multi-tenant host implements the + protocol itself and keys off ``access`` instead.""" + + def __init__(self, memory: Memory | None) -> None: + self._memory = memory + + def memory_for_access(self, access: AccessContext) -> Memory | None: + return self._memory diff --git a/python/server/src/smooth_operator_server/server.py b/python/server/src/smooth_operator_server/server.py index 09b27d7d..aec98228 100644 --- a/python/server/src/smooth_operator_server/server.py +++ b/python/server/src/smooth_operator_server/server.py @@ -38,6 +38,7 @@ from .confirmation import ConfirmationRegistry from .dispatcher import FrameDispatcher from .interaction import InteractionRegistry, PendingInteractions +from .memory import MemoryProvider from .otp import OtpService from .session_store import InMemorySessionStore, SessionStore from .skills import DirSkillResolver, SkillResolver @@ -69,6 +70,10 @@ class ServerState: #: (``SMOOTH_SKILLS_DIR``); with neither, any ``skill`` field is a clean #: ``SKILL_NOT_FOUND``, so a multi-tenant deploy never serves host skills by accident. skill_resolver: SkillResolver | None = None + #: Supplies each turn's durable-recall store, scoped to the connection's access + #: (th-ebe27d / Rust #330). ``None`` → no auto-recall, which is every deployment + #: that has not opted in. + memory_provider: MemoryProvider | None = None model: str | None = None #: Tools the agent may call during a turn (default none). Each is an engine #: ``FunctionTool``/``Tool``; the turn runner passes them straight to the agent. @@ -172,6 +177,7 @@ async def associate(target: Target) -> None: # Mirrors the Rust/TS rule: explicit resolver wins, else the env-configured # default, else off. skill_resolver=state.skill_resolver or DirSkillResolver.from_env(), + memory_provider=state.memory_provider, model=state.model, tools=state.tools, confirm_tools=state.confirm_tools, diff --git a/python/server/src/smooth_operator_server/turn_runner.py b/python/server/src/smooth_operator_server/turn_runner.py index 4e6a98e0..eb61099b 100644 --- a/python/server/src/smooth_operator_server/turn_runner.py +++ b/python/server/src/smooth_operator_server/turn_runner.py @@ -29,6 +29,7 @@ HumanApprovalResponse, InProcessExecutor, Knowledge, + Memory, SmoothAgent, SmoothAgentThread, TextEvent, @@ -241,6 +242,7 @@ def __init__( chat_client: Any, store: SessionStore, knowledge: Knowledge | None = None, + memory: Memory | None = None, system_prompt: str | None = None, skill_section: str | None = None, model: str | None = None, @@ -265,6 +267,9 @@ def __init__( #: verbatim delegation to ``SmoothAgent.run_stream`` — behavior unchanged. self._executor = select_turn_executor(executor) self._knowledge = knowledge + #: Durable-recall store for this turn, already resolved for the connection's access + #: by the dispatcher (th-ebe27d / Rust #330). ``None`` → no auto-recall. + self._memory = memory self._system_prompt = system_prompt or DEFAULT_SYSTEM_PROMPT #: Pre-rendered `## Skill: ` section for THIS turn (th-ebe27d / Rust #: #338), already resolved by the dispatcher. Appended last in @@ -389,6 +394,11 @@ async def run( "max_tokens": DEFAULT_MAX_TOKENS, "max_iterations": DEFAULT_MAX_ITERATIONS, } + # Durable auto-recall: with a store attached the engine pulls the entries relevant to + # the user's message into context. None (every deployment that has not opted in) leaves + # the turn byte-for-byte unchanged. + if self._memory is not None: + options_kwargs["memory"] = self._memory if agent_tools: options_kwargs["tools"] = agent_tools if self._model is not None: diff --git a/python/server/tests/test_memory_provider.py b/python/server/tests/test_memory_provider.py new file mode 100644 index 00000000..3396e11b --- /dev/null +++ b/python/server/tests/test_memory_provider.py @@ -0,0 +1,116 @@ +"""Durable auto-recall parity (th-ebe27d / Rust PR #330 — the +``StorageAdapter::memory_for_access`` seam, tested in +``rust/smooth-operator-server/tests/injection_seams.rs``). + +The engine already knew how to recall; what was missing on this server was the host's way +to say WHICH store, so every turn ran without auto-recall no matter what the deployment had. +Tests are named after their Rust counterparts so a parity gap stays visible. + +The recall block's header text is deliberately NOT asserted: the five cores currently inject +three different strings for it (th-ffaeae). The assertion is on the recalled CONTENT reaching +the model, which is the behavior the seam exists for. +""" + +from __future__ import annotations + +import json + +import pytest +from smooth_operator_core import InMemoryMemory, MockLlmProvider + +from smooth_operator_server.agent_config import StaticAgentConfigResolver +from smooth_operator_server.auth import AccessContext +from smooth_operator_server.dispatcher import FrameDispatcher +from smooth_operator_server.memory import StaticMemoryProvider +from smooth_operator_server.session_store import InMemorySessionStore + +RECALLED = "always add shows to the smoo-hub watchlist" + + +def _all_content_seen(chat: MockLlmProvider) -> str: + """Everything the model was sent this turn, flattened — the surface a recalled memory + must show up in.""" + return json.dumps([c.messages for c in chat.calls]) + + +async def _run_turn(provider, message: str) -> MockLlmProvider: + store = InMemorySessionStore() + session = await store.create_session("agent-x", None, None) + chat = MockLlmProvider() + chat.push_text("ok") + dispatcher = FrameDispatcher( + store, + chat, + tools=[], + agent_config_resolver=StaticAgentConfigResolver({}), + memory_provider=provider, + ) + await dispatcher.dispatch( + json.dumps({"action": "send_message", "sessionId": session.session_id, "message": message}), + lambda _e: None, + ) + await dispatcher.wait_for_turns() + return chat + + +def _memory_with_entry() -> InMemoryMemory: + memory = InMemoryMemory() + memory.remember(RECALLED) + return memory + + +# ── rust: no_memory_means_no_recall_injection ──────────────────────────────── + + +@pytest.mark.asyncio +async def test_no_memory_means_no_recall_injection() -> None: + """Default: no provider ⇒ no auto-recall. Guards against the seam injecting when absent — + an unopted deployment's turn must be byte-for-byte what it was before.""" + chat = await _run_turn(None, "add shows to my watchlist") + assert RECALLED not in _all_content_seen(chat) + + +@pytest.mark.asyncio +async def test_provider_returning_none_means_no_recall_injection() -> None: + """A provider that returns None for this caller is the same as no provider — the seam must + not fabricate a store just because one was installed.""" + chat = await _run_turn(StaticMemoryProvider(None), "add shows to my watchlist") + assert RECALLED not in _all_content_seen(chat) + + +# ── rust: attached_memory_is_auto_recalled_into_the_turn ───────────────────── + + +@pytest.mark.asyncio +async def test_attached_memory_is_auto_recalled_into_the_turn() -> None: + """With a store attached the engine recalls the entries relevant to the user's message and + injects them into the turn — the seam that lights up Big Smooth's durable auto-recall.""" + # The message shares "add", "shows", "watchlist" with the stored entry, so the engine's + # word-overlap recall surfaces it. + chat = await _run_turn(StaticMemoryProvider(_memory_with_entry()), "add shows to my watchlist") + assert RECALLED in _all_content_seen(chat) + + +@pytest.mark.asyncio +async def test_irrelevant_message_recalls_nothing() -> None: + """An unrelated message recalls nothing: the seam is relevance-gated by the engine, not a + blanket dump of every stored memory into every turn. The message shares NO token with the + entry — the bundled lexical scorer counts raw token overlap with no stopword filter, so a + single shared "the" would be enough to score a hit.""" + chat = await _run_turn(StaticMemoryProvider(_memory_with_entry()), "explain quantum entanglement") + assert RECALLED not in _all_content_seen(chat) + + +@pytest.mark.asyncio +async def test_provider_sees_the_callers_access() -> None: + """The seam is access-scoped (mirroring knowledge) so a multi-tenant host can bind memory to + the requester — the argument must actually reach the provider.""" + seen: list[AccessContext] = [] + + class _Recording: + def memory_for_access(self, access: AccessContext): + seen.append(access) + return None + + await _run_turn(_Recording(), "hello") + assert len(seen) == 1 diff --git a/python/uv.lock b/python/uv.lock index a480d6c6..f74a277f 100644 --- a/python/uv.lock +++ b/python/uv.lock @@ -775,7 +775,7 @@ wheels = [ [[package]] name = "smooai-smooth-operator" -version = "1.51.0" +version = "1.58.7" source = { editable = "." } dependencies = [ { name = "jsonschema" }, diff --git a/typescript/server/src/frameDispatcher.ts b/typescript/server/src/frameDispatcher.ts index 71404a79..42725b7d 100644 --- a/typescript/server/src/frameDispatcher.ts +++ b/typescript/server/src/frameDispatcher.ts @@ -16,6 +16,7 @@ import type { StoredSession } from './sessionStore.js'; import { randomUUID } from 'node:crypto'; import { type AgentConfigResolver, assembleSystemPrompt } from './agentConfig.js'; +import type { MemoryProvider } from './memory.js'; import { resolveSection, type SkillResolver } from './skills.js'; import { gateTools, type SessionAuthenticator } from './toolGating.js'; import { ANONYMOUS_ACCESS, type AccessContext } from './auth.js'; @@ -53,6 +54,8 @@ export interface FrameDispatcherOptions { * `SKILL_NOT_FOUND`, so a multi-tenant deploy never serves host skills by accident. */ skillResolver?: SkillResolver; + /** Supplies the turn's durable-recall store, scoped to this connection's access (Rust #330). */ + memoryProvider?: MemoryProvider; /** Tools the agent may call during a turn (default none); forwarded to the {@link TurnRunner}. */ tools?: Tool[]; /** @@ -136,6 +139,7 @@ export class FrameDispatcher { associate?: (target: Target) => void; private readonly systemPrompt?: string; private readonly skillResolver?: SkillResolver; + private readonly memoryProvider?: MemoryProvider; private readonly tools: Tool[]; private readonly toolHooks: ToolHook[]; private readonly confirmTools: string[]; @@ -168,6 +172,7 @@ export class FrameDispatcher { this.access = options.access ?? ANONYMOUS_ACCESS; this.systemPrompt = options.systemPrompt; this.skillResolver = options.skillResolver; + this.memoryProvider = options.memoryProvider; this.tools = options.tools ?? []; this.toolHooks = options.toolHooks ?? []; this.confirmTools = options.confirmTools ?? []; @@ -723,6 +728,10 @@ export class FrameDispatcher { chatClient: this.chatClient, store: this.store, knowledge: scopedKnowledge, + // Durable auto-recall, scoped the way retrieval is: the host decides which memory this + // caller recalls from. Absent (no provider, or one that declines this access) leaves the + // turn without auto-recall — byte-for-byte unchanged. + memory: this.memoryProvider?.memoryForAccess(this.access), systemPrompt: effectiveSystemPrompt, tools: effectiveTools, toolHooks: this.toolHooks, diff --git a/typescript/server/src/index.ts b/typescript/server/src/index.ts index 5be5ee89..fbee4dce 100644 --- a/typescript/server/src/index.ts +++ b/typescript/server/src/index.ts @@ -37,6 +37,7 @@ export type { IntakeField, IntakeFieldError, IntakeFieldKey, IntakeValues } from export { DEFAULT_MAX_ITERATIONS, DEFAULT_MAX_TOKENS, DEFAULT_MODEL, DEFAULT_SYSTEM_PROMPT, TurnRunner } from './turnRunner.js'; export type { Sink, TurnResult, TurnRunnerOptions } from './turnRunner.js'; +export { StaticMemoryProvider, type MemoryProvider } from './memory.js'; export { DirSkillResolver, isValidSkillName, resolveSection, skillSection, stripFrontmatter, SKILLS_DIR_ENV, type SkillResolver } from './skills.js'; export { assembleSystemPrompt, parseAgentConfig, StaticAgentConfigResolver } from './agentConfig.js'; export type { AgentConfig, AgentConfigResolver, EnabledTool } from './agentConfig.js'; diff --git a/typescript/server/src/memory.ts b/typescript/server/src/memory.ts new file mode 100644 index 00000000..c9df3e4b --- /dev/null +++ b/typescript/server/src/memory.ts @@ -0,0 +1,38 @@ +/** + * Durable auto-recall — the server side of the engine's `Memory` seam (Rust PR #330). + * + * The engine already knows how to auto-recall: give `AgentOptions.memory` a store and it pulls the + * entries relevant to the user's message into the turn's context. What was missing on this server is + * the way for a HOST to say *which* store — so every turn ran without auto-recall regardless of what + * the deployment had. + * + * Mirrors the Rust `StorageAdapter::memory_for_access` seam. `access` is threaded (as it is for + * knowledge) so a multi-tenant backend can bind memory to the requester's org/user; a single-tenant + * host — Big Smooth's daemon, the reason this seam exists — ignores it and returns its one store. + */ +import type { Memory } from '@smooai/smooth-operator-core'; + +import type { AccessContext } from './auth.js'; + +/** Supplies the durable-recall handle for a turn. */ +export interface MemoryProvider { + /** + * The memory to auto-recall from for a caller with this access, or `undefined` for none. + * `undefined` is the default for every deployment that has not opted in, and leaves the turn + * byte-for-byte unchanged. + */ + memoryForAccess(access: AccessContext): Memory | undefined; +} + +/** + * A {@link MemoryProvider} over one unscoped store — the single-tenant case (Big Smooth's daemon + * hands its SQLite-backed store straight through). A multi-tenant host implements the interface + * itself and keys off `access` instead. + */ +export class StaticMemoryProvider implements MemoryProvider { + constructor(private readonly memory: Memory | undefined) {} + + memoryForAccess(_access: AccessContext): Memory | undefined { + return this.memory; + } +} diff --git a/typescript/server/src/server.ts b/typescript/server/src/server.ts index 3e8c1f8d..1e19e4b4 100644 --- a/typescript/server/src/server.ts +++ b/typescript/server/src/server.ts @@ -27,6 +27,7 @@ import type { AgentConfigResolver } from './agentConfig.js'; import type { SessionAuthenticator } from './toolGating.js'; import type { OtpService } from './otp.js'; import { type AccessKnowledge, FrameDispatcher } from './frameDispatcher.js'; +import type { MemoryProvider } from './memory.js'; import { DirSkillResolver, type SkillResolver } from './skills.js'; import type { ModelCeilingResolver } from './modelCeiling.js'; import type { ToolContext, ToolProvider } from './toolContext.js'; @@ -120,6 +121,12 @@ export interface ServerOptions { * `SKILL_NOT_FOUND`, so a multi-tenant deploy never serves host skills by accident. */ skillResolver?: SkillResolver; + + /** + * Supplies each turn's durable-recall store, scoped to the connection's access (th-ebe27d / + * Rust #330). Omitted ⇒ no auto-recall, which is every deployment that has not opted in. + */ + memoryProvider?: MemoryProvider; /** * The Rich Interactions the server hosts (see `interaction.ts`). Each turn registers * one `request_` raise tool per kind, gated per-kind by the session's declared @@ -206,6 +213,7 @@ export function buildServer(options: ServerOptions): { toolProvider: options.toolProvider, // Mirrors Rust's install_skill_resolver_from_env: explicit wins, else env, else off. skillResolver: options.skillResolver ?? DirSkillResolver.fromEnv(), + memoryProvider: options.memoryProvider, }); // Fire-and-forget the per-connection loop; it owns the socket's lifecycle. void runConnection(socket, dispatcher, backplane, drain.signal); diff --git a/typescript/server/src/turnRunner.ts b/typescript/server/src/turnRunner.ts index 5dd22f49..1ea6af2f 100644 --- a/typescript/server/src/turnRunner.ts +++ b/typescript/server/src/turnRunner.ts @@ -12,7 +12,7 @@ * `runStream` mapped event-by-event onto protocol events. */ import { approve, deny, SmoothAgent } from '@smooai/smooth-operator-core'; -import type { AgentExecutor, AgentOptions, ChatClientLike, HumanApprovalRequest, HumanApprovalResponse, Knowledge, StreamEvent, Tool, ToolHook } from '@smooai/smooth-operator-core'; +import type { AgentExecutor, AgentOptions, ChatClientLike, HumanApprovalRequest, HumanApprovalResponse, Knowledge, Memory, StreamEvent, Tool, ToolHook } from '@smooai/smooth-operator-core'; import type { ConfirmationRegistry } from './confirmation.js'; import { turnExecutor } from './executorSelection.js'; @@ -166,6 +166,11 @@ export interface TurnRunnerOptions { store: SessionStore; /** Optional knowledge retriever, already SCOPED to the connection's access (ACL). */ knowledge?: Knowledge; + /** + * Durable-recall store for this turn, already resolved for the connection's access by the + * dispatcher (th-ebe27d / Rust #330). Absent → the turn runs without auto-recall. + */ + memory?: Memory; systemPrompt?: string; /** Tools the agent may call during the turn (default none); passed straight to the engine. */ tools?: Tool[]; @@ -245,6 +250,7 @@ export class TurnRunner { private readonly chatClient: ChatClientLike; private readonly store: SessionStore; private readonly knowledge?: Knowledge; + private readonly memory?: Memory; private readonly systemPrompt: string; private readonly tools: Tool[]; private readonly toolHooks: ToolHook[]; @@ -271,6 +277,7 @@ export class TurnRunner { this.chatClient = options.chatClient; this.store = options.store; this.knowledge = options.knowledge; + this.memory = options.memory; this.systemPrompt = options.systemPrompt ?? DEFAULT_SYSTEM_PROMPT; this.tools = options.tools ?? []; this.toolHooks = options.toolHooks ?? []; @@ -346,6 +353,10 @@ export class TurnRunner { maxIterations: DEFAULT_MAX_ITERATIONS, }; if (this.knowledge) agentOptions.knowledge = this.knowledge; + // Durable auto-recall: with a store attached the engine pulls the entries relevant to the + // user's message into context. Absent (every deployment that has not opted in) leaves the + // turn byte-for-byte unchanged. + if (this.memory) agentOptions.memory = this.memory; if (this.tools.length > 0) agentOptions.tools = this.tools; // Thread consumer-supplied surveillance hooks into the engine's per-turn tool // registry. Empty ⇒ unset ⇒ behaviour unchanged. diff --git a/typescript/server/test/memoryProvider.test.ts b/typescript/server/test/memoryProvider.test.ts new file mode 100644 index 00000000..5a4d7ece --- /dev/null +++ b/typescript/server/test/memoryProvider.test.ts @@ -0,0 +1,100 @@ +/** + * Durable auto-recall (Rust PR #330 — the `StorageAdapter::memory_for_access` seam, tested in + * `rust/smooth-operator-server/tests/injection_seams.rs`) — the TS server's parity. + * + * The engine already knew how to recall; what was missing on this server was the host's way to say + * WHICH store, so every turn ran without auto-recall no matter what the deployment had. Tests are + * named after their Rust counterparts so a parity gap stays visible. + * + * The recall block's header text is deliberately NOT asserted: the five cores currently inject three + * different strings for it (th-ffaeae). The assertion is on the recalled CONTENT reaching the model, + * which is the behavior the seam exists for. + */ +import { InMemoryMemory, MockLlmProvider } from '@smooai/smooth-operator-core'; +import { afterEach, describe, expect, it } from 'vitest'; + +import type { AccessContext } from '../src/auth.js'; +import { StaticMemoryProvider, type MemoryProvider } from '../src/memory.js'; +import { serve, type RunningServer } from '../src/server.js'; +import { TestClient } from './wsClient.js'; + +const RECALLED = 'always add shows to the smoo-hub watchlist'; + +/** Everything the model was sent this turn, flattened — the surface a recalled memory must show up in. */ +function allContentSeen(chat: MockLlmProvider): string { + return JSON.stringify(chat.calls.map((c) => c.messages)); +} + +function memoryWithEntry(): InMemoryMemory { + const memory = new InMemoryMemory(); + memory.remember(RECALLED); + return memory; +} + +describe('durable auto-recall (over the wire)', () => { + let server: RunningServer | undefined; + afterEach(async () => { + await server?.close(); + server = undefined; + }); + + async function runTurn(memoryProvider: MemoryProvider | undefined, message: string): Promise { + const chat = new MockLlmProvider().pushText('ok'); + server = await serve({ chatClient: chat, memoryProvider }); + const client = await TestClient.connect(server.url); + client.sendAction({ action: 'create_conversation_session', requestId: 'cs', agentId: 'agent' }); + const sessionId = ((await client.receive()).data as Record).sessionId as string; + + client.sendAction({ action: 'send_message', requestId: 'r2', sessionId, message }); + await client.receiveUntil('eventual_response'); + await client.close(); + return chat; + } + + // ── rust: no_memory_means_no_recall_injection ──────────────────────────── + + it('injects no recall when no provider is installed', async () => { + // Guards against the seam injecting when absent — an unopted deployment's turn must be + // byte-for-byte what it was before. + const chat = await runTurn(undefined, 'add shows to my watchlist'); + expect(allContentSeen(chat)).not.toContain(RECALLED); + }); + + it('injects no recall when the provider declines this caller', async () => { + // A provider returning undefined is the same as no provider — the seam must not fabricate a + // store just because one was installed. + const chat = await runTurn(new StaticMemoryProvider(undefined), 'add shows to my watchlist'); + expect(allContentSeen(chat)).not.toContain(RECALLED); + }); + + // ── rust: attached_memory_is_auto_recalled_into_the_turn ───────────────── + + it('auto-recalls an attached memory into the turn', async () => { + // The message shares "add", "shows", "watchlist" with the stored entry, so the engine's + // word-overlap recall surfaces it. + const chat = await runTurn(new StaticMemoryProvider(memoryWithEntry()), 'add shows to my watchlist'); + expect(allContentSeen(chat)).toContain(RECALLED); + }); + + it('recalls nothing for an unrelated message', async () => { + // Relevance-gated by the engine, not a blanket dump of every stored memory into every turn. + // The message shares NO token with the entry — the bundled lexical scorer counts raw token + // overlap with no stopword filter, so a single shared "the" would be enough to score a hit. + const chat = await runTurn(new StaticMemoryProvider(memoryWithEntry()), 'explain quantum entanglement'); + expect(allContentSeen(chat)).not.toContain(RECALLED); + }); + + it('hands the provider the caller access, so a multi-tenant host can scope by requester', async () => { + const seen: AccessContext[] = []; + await runTurn( + { + memoryForAccess(access: AccessContext) { + seen.push(access); + return undefined; + }, + }, + 'hello', + ); + expect(seen).toHaveLength(1); + }); +});