diff --git a/.changeset/memory-provider-seam.md b/.changeset/memory-provider-seam.md new file mode 100644 index 00000000..54668f32 --- /dev/null +++ b/.changeset/memory-provider-seam.md @@ -0,0 +1,46 @@ +--- +"@smooai/smooth-operator": patch +--- + +feat(dotnet,python,ts,go): let a host supply the turn's memory — durable auto-recall off Rust (th-ebe27d) + +Rust #330 put `memory_for_access` on `StorageAdapter` and had the server runner thread the +result into the engine's agent options, which is what lights up Big Smooth's durable +auto-recall. The four sibling servers never did — and the gap was invisible, because **all +five engine cores already implement `Memory` and already recall relevant entries into +context**. The capability was fully built on both ends with nothing connecting them: no +matter what store a deployment had, every turn on these servers ran without auto-recall. + +Each server now takes a `MemoryProvider` (`IMemoryProvider` in C#) with one method — +`memory_for_access(access)` — resolved per turn and passed to the engine as +`AgentOptions.memory`: + +| | seam | install | +|---|---|---| +| C# | `IMemoryProvider` | DI (`services.GetService()`) | +| Python | `MemoryProvider` | `ServerState.memory_provider` | +| TypeScript | `MemoryProvider` | `serve({ memoryProvider })` | +| Go | `MemoryProvider` | `WithMemoryProvider(...)` | + +`access` is threaded exactly as it is for knowledge, so a multi-tenant host can bind memory +to the requester's org/user; single-tenant hosts — Big Smooth's daemon, the reason the seam +exists — ignore it, so each language also ships a `StaticMemoryProvider` over one store. + +**Nothing changes for anyone who does not opt in.** No provider, or a provider that returns +nothing for this caller, leaves the turn byte-for-byte what it was — and that is a test, not +a claim: each language asserts the no-provider and the declining-provider paths inject +nothing, alongside the positive case and a relevance case (an unrelated message recalls +nothing, so this is not a blanket dump of every stored memory into every turn). + +Five tests per language, named after their Rust counterparts in +`rust/smooth-operator-server/tests/injection_seams.rs`, all four mutation-checked — dropping +the one line that hands memory to the engine fails them. + +Two notes for whoever picks this up next. The recall block's **header text is deliberately +not asserted**: the five cores currently inject three different strings for it (th-ffaeae), +so the tests assert the recalled *content* reaches the model, which is the behavior the seam +exists for. And the bundled lexical scorer counts raw token overlap with **no stopword +filter**, so in practice a single shared "the" scores a hit — worth knowing before trusting +recall precision in production. + +No wire-protocol change. diff --git a/dotnet/server/aspnetcore/SmoothOperatorWebSocketExtensions.cs b/dotnet/server/aspnetcore/SmoothOperatorWebSocketExtensions.cs index 1164efb9..5af7de19 100644 --- a/dotnet/server/aspnetcore/SmoothOperatorWebSocketExtensions.cs +++ b/dotnet/server/aspnetcore/SmoothOperatorWebSocketExtensions.cs @@ -123,7 +123,10 @@ private static FrameDispatcher BuildDispatcher(HttpContext context) // fall back to the env-configured directory resolver (SMOOTH_SKILLS_DIR). Unset ⇒ null ⇒ any // `skill` field is a clean SKILL_NOT_FOUND, so a multi-tenant deploy never serves host skills // by accident. Mirrors Rust's install_skill_resolver_from_env. - skillResolver: services.GetService() ?? DirSkillResolver.FromEnv()); + skillResolver: services.GetService() ?? DirSkillResolver.FromEnv(), + // Durable auto-recall (th-ebe27d / Rust #330). A host registers an IMemoryProvider — or a + // StaticMemoryProvider over one store — to light it up; unregistered ⇒ no auto-recall. + memoryProvider: services.GetService()); } private static async Task PumpAsync( diff --git a/dotnet/server/src/FrameDispatcher.cs b/dotnet/server/src/FrameDispatcher.cs index b16b73af..5a1b8905 100644 --- a/dotnet/server/src/FrameDispatcher.cs +++ b/dotnet/server/src/FrameDispatcher.cs @@ -59,6 +59,9 @@ public sealed class FrameDispatcher private readonly TurnLimits _limits; private readonly ILogger? _logger; private readonly ISkillResolver? _skillResolver; + /// Supplies the turn's durable-recall store, scoped to this connection's access + /// (th-ebe27d / Rust #330). Null → no auto-recall, the default. + private readonly IMemoryProvider? _memoryProvider; // The connection's SINGLE in-flight send_message turn, if one is running. A turn that calls a // confirmation-gated tool parks awaiting a later confirm_tool_action frame, so the turn runs as a @@ -102,7 +105,8 @@ public FrameDispatcher( TurnLimits? limits = null, ILogger? logger = null, IReadOnlyList? toolHooks = null, - ISkillResolver? skillResolver = null) + ISkillResolver? skillResolver = null, + IMemoryProvider? memoryProvider = null) { _store = store ?? throw new ArgumentNullException(nameof(store)); _chatClient = chatClient ?? throw new ArgumentNullException(nameof(chatClient)); @@ -110,6 +114,7 @@ public FrameDispatcher( _access = access ?? AccessContext.Anonymous; _systemPrompt = systemPrompt; _skillResolver = skillResolver; + _memoryProvider = memoryProvider; _reranker = reranker; _tools = tools ?? Array.Empty(); // Tool-call hooks (surveillance / redaction) forwarded to every turn's registry (empty → no @@ -784,6 +789,11 @@ private async Task HandleSendMessageAsync(JsonObject frame, string? requestId, A // through the same ACL-filtered store (a doc the caller's groups don't grant is never a candidate). var scopedKnowledge = _knowledge?.ForAccess(_access); + // Durable auto-recall, scoped the same way retrieval is: the host decides which memory this + // caller recalls from. Null (no provider installed, or one that declines this access) leaves + // the turn without auto-recall — byte-for-byte unchanged. + var scopedMemory = _memoryProvider?.MemoryForAccess(_access); + // Built-in knowledge_search: a model-callable search over the connection's ACL-scoped knowledge // (parity with the Rust server's KnowledgeSearchTool). Prepended before the enabled_tools filter // so it flows through the SAME per-agent restriction + auth gate as every other tool — an agent @@ -824,7 +834,7 @@ private async Task HandleSendMessageAsync(JsonObject frame, string? requestId, A // have arrived on a PREVIOUS connection (a reconnect resumes the conversation on a fresh // dispatcher), and a per-connection map would read empty there. th-13df6d. var capabilities = await _store.GetClientSupportsAsync(session.ConversationId, cancellationToken).ConfigureAwait(false); - var runner = new TurnRunner(_chatClient, _store, scopedKnowledge, _systemPrompt, _reranker, gatedTools, confirmTools, _confirmations, agentConfig, _judge, _limits, _logger, toolHooks: _toolHooks, interactions: _interactions, interactionPark: _interactionPark, capabilities: capabilities, interactionEffects: _sessionIdentity) + var runner = new TurnRunner(_chatClient, _store, scopedKnowledge, _systemPrompt, _reranker, gatedTools, confirmTools, _confirmations, agentConfig, _judge, _limits, _logger, toolHooks: _toolHooks, interactions: _interactions, interactionPark: _interactionPark, capabilities: capabilities, interactionEffects: _sessionIdentity, memory: scopedMemory) { ConfirmationTimeout = ConfirmationTimeout, }; diff --git a/dotnet/server/src/Memory.cs b/dotnet/server/src/Memory.cs new file mode 100644 index 00000000..bb1c5fd5 --- /dev/null +++ b/dotnet/server/src/Memory.cs @@ -0,0 +1,42 @@ +using SmooAI.SmoothOperator.Core; + +namespace SmooAI.SmoothOperator.Server; + +/// +/// Supplies the durable-recall handle for a turn — the C# analog of the Rust +/// StorageAdapter::memory_for_access seam (PR #330). +/// +/// The engine already knows how to auto-recall: give AgentOptions.Memory a store and it +/// pulls the entries relevant to the user's message into the turn's context. What was missing on +/// this server is the way for a HOST to say which store — so every turn ran without +/// auto-recall regardless of what the deployment had. +/// +/// The access argument is threaded (mirroring IKnowledgeBase.ForAccess) so a +/// multi-tenant backend can bind memory to the requester's org/user; a single-tenant host — Big +/// Smooth's daemon, which is the reason this seam exists — ignores it and returns its one store. +/// +public interface IMemoryProvider +{ + /// + /// The memory to auto-recall from for a caller with this access, or null for none. + /// null is the default for every deployment that has not opted in, and leaves the turn + /// byte-for-byte unchanged. + /// + IAgentMemory? MemoryForAccess(AccessContext access); +} + +/// +/// An over one unscoped store — the single-tenant case (Big Smooth's +/// daemon hands its SQLite-backed store straight through). A multi-tenant host implements the +/// interface itself and keys off access instead. +/// +public sealed class StaticMemoryProvider : IMemoryProvider +{ + private readonly IAgentMemory? _memory; + + /// The store every caller recalls from; null disables auto-recall. + public StaticMemoryProvider(IAgentMemory? memory) => _memory = memory; + + /// + public IAgentMemory? MemoryForAccess(AccessContext access) => _memory; +} diff --git a/dotnet/server/src/TurnRunner.cs b/dotnet/server/src/TurnRunner.cs index 0724ba76..87a2165e 100644 --- a/dotnet/server/src/TurnRunner.cs +++ b/dotnet/server/src/TurnRunner.cs @@ -62,6 +62,9 @@ public sealed class TurnRunner private readonly IChatClient _chatClient; private readonly ISessionStore _store; private readonly IKnowledgeBase? _knowledge; + /// Durable-recall store for this turn, already resolved for the connection's access by + /// the dispatcher (th-ebe27d / Rust #330). Null → the turn runs without auto-recall. + private readonly IAgentMemory? _memory; private readonly IReranker? _reranker; private readonly string _systemPrompt; private readonly IReadOnlyList _tools; @@ -96,11 +99,12 @@ public sealed class TurnRunner /// — or a test — narrows it). public TimeSpan ConfirmationTimeout { get; init; } = DefaultConfirmationTimeout; - public TurnRunner(IChatClient chatClient, ISessionStore store, IKnowledgeBase? knowledge = null, string? systemPrompt = null, IReranker? reranker = null, IReadOnlyList? tools = null, IReadOnlyList? confirmTools = null, ConfirmationRegistry? confirmations = null, AgentConfig? agentConfig = null, IWorkflowJudge? judge = null, TurnLimits? limits = null, ILogger? logger = null, IChatClient? preambleChatClient = null, IReadOnlyList? toolHooks = null, InteractionCatalog? interactions = null, InteractionParkRegistry? interactionPark = null, IReadOnlyCollection? capabilities = null, SessionIdentityRegistry? interactionEffects = null) + public TurnRunner(IChatClient chatClient, ISessionStore store, IKnowledgeBase? knowledge = null, string? systemPrompt = null, IReranker? reranker = null, IReadOnlyList? tools = null, IReadOnlyList? confirmTools = null, ConfirmationRegistry? confirmations = null, AgentConfig? agentConfig = null, IWorkflowJudge? judge = null, TurnLimits? limits = null, ILogger? logger = null, IChatClient? preambleChatClient = null, IReadOnlyList? toolHooks = null, InteractionCatalog? interactions = null, InteractionParkRegistry? interactionPark = null, IReadOnlyCollection? capabilities = null, SessionIdentityRegistry? interactionEffects = null, IAgentMemory? memory = null) { _chatClient = chatClient ?? throw new ArgumentNullException(nameof(chatClient)); _store = store ?? throw new ArgumentNullException(nameof(store)); _knowledge = knowledge; + _memory = memory; _reranker = reranker; _systemPrompt = systemPrompt ?? "You are a helpful customer support agent. Answer using only the knowledge provided to you; if it is not there, say you don't know."; @@ -331,6 +335,10 @@ public async Task RunAsync(string conversationId, string requestId, MaxIterations = _limits.MaxIterations, MaxOutputTokens = _limits.MaxTokens, ModelMaxOutputTokens = _limits.ModelMaxOutputTokens, + // Durable auto-recall: with a store attached the engine pulls the entries relevant to the + // user's message into context. Null (every deployment that has not opted in) leaves the + // turn byte-for-byte unchanged. + Memory = _memory, }; foreach (var tool in _tools) { diff --git a/dotnet/server/tests/MemoryProviderTests.cs b/dotnet/server/tests/MemoryProviderTests.cs new file mode 100644 index 00000000..bb7ab73e --- /dev/null +++ b/dotnet/server/tests/MemoryProviderTests.cs @@ -0,0 +1,124 @@ +using System.Text.Json.Nodes; +using SmooAI.SmoothOperator.Core; +using SmooAI.SmoothOperator.Server; + +namespace SmooAI.SmoothOperator.Server.Tests; + +/// +/// Durable auto-recall parity with the Rust reference (PR #330 — the +/// StorageAdapter::memory_for_access seam, tested in +/// rust/smooth-operator-server/tests/injection_seams.rs). +/// +/// The engine already knew how to recall; what was missing on this server was the host's way to say +/// WHICH store, so every turn ran without auto-recall no matter what the deployment had. These tests +/// are named after their Rust counterparts so a parity gap stays visible. +/// +/// The recall block's header text is deliberately NOT asserted here: the five cores currently inject +/// three different strings for it (th-ffaeae). The assertion is on the recalled CONTENT reaching the +/// model, which is the behavior the seam exists for. +/// +public class MemoryProviderTests +{ + private static async Task CreateSessionAsync(FrameDispatcher dispatcher, List events) + { + await dispatcher.DispatchAsync("""{"action":"create_conversation_session","agentId":"11111111-1111-1111-1111-111111111111","requestId":"r1"}""", events.Add); + var sessionId = events[0]["data"]!["sessionId"]!.GetValue(); + events.Clear(); + return sessionId; + } + + /// Everything the model was sent this turn, flattened — the surface a recalled memory + /// must show up in. + private static string AllContentSeen(RecordingChatClient chat) => + string.Join("\n", chat.LastMessages.Select(m => m.Text)); + + private static async Task RunTurnAsync(IMemoryProvider? provider, string message) + { + var chat = new RecordingChatClient("ok"); + var dispatcher = new FrameDispatcher(new InMemorySessionStore(), chat, memoryProvider: provider); + var events = new List(); + var sessionId = await CreateSessionAsync(dispatcher, events); + + await dispatcher.DispatchAsync( + $$"""{"action":"send_message","requestId":"r2","sessionId":"{{sessionId}}","message":"{{message}}"}""", + events.Add); + await dispatcher.WaitForTurnsAsync(); + return chat; + } + + // ── rust: no_memory_means_no_recall_injection ──────────────────────────── + + /// Default: no provider ⇒ no auto-recall. Guards against the seam injecting when absent — + /// an unopted deployment's turn must be byte-for-byte what it was before. + [Fact] + public async Task NoMemoryMeansNoRecallInjection() + { + var chat = await RunTurnAsync(null, "add shows to my watchlist"); + Assert.DoesNotContain("smoo-hub watchlist", AllContentSeen(chat), StringComparison.Ordinal); + } + + /// A provider that returns null for this caller is the same as no provider — the seam + /// must not fabricate a store just because one was installed. + [Fact] + public async Task ProviderReturningNullMeansNoRecallInjection() + { + var chat = await RunTurnAsync(new StaticMemoryProvider(null), "add shows to my watchlist"); + Assert.DoesNotContain("smoo-hub watchlist", AllContentSeen(chat), StringComparison.Ordinal); + } + + // ── rust: attached_memory_is_auto_recalled_into_the_turn ───────────────── + + /// With a store attached the engine recalls the entries relevant to the user's message + /// and injects them into the turn — the seam that lights up Big Smooth's durable auto-recall. + [Fact] + public async Task AttachedMemoryIsAutoRecalledIntoTheTurn() + { + var memory = new InMemoryAgentMemory(); + await memory.StoreAsync(new MemoryEntry("m-1", "always add shows to the smoo-hub watchlist", MemoryType.Project)); + + // The message shares "add", "shows", "watchlist" with the stored entry, so the engine's + // word-overlap recall surfaces it. + var chat = await RunTurnAsync(new StaticMemoryProvider(memory), "add shows to my watchlist"); + + Assert.Contains("smoo-hub watchlist", AllContentSeen(chat), StringComparison.Ordinal); + } + + /// An unrelated message recalls nothing: the seam is relevance-gated by the engine, not a + /// blanket dump of every stored memory into every turn. The message shares NO token with the entry — + /// the bundled lexical scorer counts raw token overlap with no stopword filter, so a single shared + /// "the" is enough to score a hit. + [Fact] + public async Task IrrelevantMessageRecallsNothing() + { + var memory = new InMemoryAgentMemory(); + await memory.StoreAsync(new MemoryEntry("m-1", "always add shows to the smoo-hub watchlist", MemoryType.Project)); + + var chat = await RunTurnAsync(new StaticMemoryProvider(memory), "explain quantum entanglement"); + + Assert.DoesNotContain("smoo-hub watchlist", AllContentSeen(chat), StringComparison.Ordinal); + } + + /// The seam is access-scoped (mirroring IKnowledgeBase.ForAccess) so a multi-tenant + /// host can bind memory to the requester — the argument must actually reach the provider. + [Fact] + public async Task ProviderSeesTheCallersAccess() + { + var seen = new List(); + var chat = await RunTurnAsync(new RecordingMemoryProvider(seen), "hello"); + + Assert.Single(seen); + } + + private sealed class RecordingMemoryProvider : IMemoryProvider + { + private readonly List _seen; + + public RecordingMemoryProvider(List seen) => _seen = seen; + + public IAgentMemory? MemoryForAccess(AccessContext access) + { + _seen.Add(access); + return null; + } + } +} diff --git a/go/server/dispatcher.go b/go/server/dispatcher.go index 6182d526..7785744f 100644 --- a/go/server/dispatcher.go +++ b/go/server/dispatcher.go @@ -39,6 +39,10 @@ type FrameDispatcher struct { // is long enough. Nil → the feature is off and any skill field is a clean // SKILL_NOT_FOUND, so a multi-tenant deploy never serves host skills by accident. skills SkillResolver + // memoryProvider supplies the turn's durable-recall store, scoped to this connection's + // access (th-ebe27d / Rust #330). Set by the server after construction, alongside skills. + // nil → no auto-recall, the default. + memoryProvider MemoryProvider // associate records this connection's backplane targets as they are learned (set by // the connection loop). Nil → session/agent targets are never routable. associate func(target Target) @@ -790,6 +794,12 @@ func (d *FrameDispatcher) handleSendMessage(ctx context.Context, frame inboundFr }() runner := NewTurnRunner(d.client, d.store, effectiveSystemPrompt, d.knowledge, effectiveTools, d.confirmTools, d.confirmations, workflow, session.CurrentStepID, d.judgeModel, d.modelCeiling) runner.hooks = d.hooks + // Durable auto-recall, scoped the way retrieval is: the host decides which memory this + // caller recalls from. nil (no provider, or one that declines this access) leaves the turn + // without auto-recall — byte-for-byte unchanged. + if d.memoryProvider != nil { + runner.memory = d.memoryProvider.MemoryForAccess(d.access) + } // Rich Interactions: give the runner the hosted kinds, the park/resume registry, // and this session's declared capabilities, so it registers one raise tool per // kind (rich park when the capability is declared, else conversational fallback). diff --git a/go/server/memory.go b/go/server/memory.go new file mode 100644 index 00000000..78d9efb2 --- /dev/null +++ b/go/server/memory.go @@ -0,0 +1,39 @@ +package server + +import ( + core "github.com/SmooAI/smooth-operator-core/go/core" +) + +// Durable auto-recall — the server side of the engine's Memory seam (Rust PR #330). +// +// The engine already knows how to auto-recall: give AgentOptions.Memory a store and it pulls the +// entries relevant to the user's message into the turn's context. What was missing on this server +// is the way for a HOST to say WHICH store — so every turn ran without auto-recall regardless of +// what the deployment had. +// +// Mirrors the Rust StorageAdapter::memory_for_access seam. access is threaded (as it is for +// knowledge) so a multi-tenant backend can bind memory to the requester's org/user; a single-tenant +// host — Big Smooth's daemon, the reason this seam exists — ignores it and returns its one store. + +// MemoryProvider supplies the durable-recall handle for a turn. +type MemoryProvider interface { + // MemoryForAccess returns the memory to auto-recall from for a caller with this access, or nil + // for none. nil is the default for every deployment that has not opted in, and leaves the turn + // byte-for-byte unchanged. + MemoryForAccess(access AccessContext) core.Memory +} + +// StaticMemoryProvider is a MemoryProvider over one unscoped store — the single-tenant case (Big +// Smooth's daemon hands its SQLite-backed store straight through). A multi-tenant host implements +// the interface itself and keys off access instead. +type StaticMemoryProvider struct { + memory core.Memory +} + +// NewStaticMemoryProvider wraps one store (nil disables auto-recall). +func NewStaticMemoryProvider(memory core.Memory) *StaticMemoryProvider { + return &StaticMemoryProvider{memory: memory} +} + +// MemoryForAccess implements MemoryProvider. +func (p *StaticMemoryProvider) MemoryForAccess(_ AccessContext) core.Memory { return p.memory } diff --git a/go/server/memory_test.go b/go/server/memory_test.go new file mode 100644 index 00000000..1dd5f305 --- /dev/null +++ b/go/server/memory_test.go @@ -0,0 +1,120 @@ +package server + +import ( + "encoding/json" + "strings" + "testing" + + core "github.com/SmooAI/smooth-operator-core/go/core" +) + +// Durable auto-recall parity (th-ebe27d / Rust PR #330 — the StorageAdapter::memory_for_access +// seam, tested in rust/smooth-operator-server/tests/injection_seams.rs). +// +// The engine already knew how to recall; what was missing on this server was the host's way to say +// WHICH store, so every turn ran without auto-recall no matter what the deployment had. Tests are +// named after their Rust counterparts so a parity gap stays visible. +// +// The recall block's header text is deliberately NOT asserted: the five cores currently inject three +// different strings for it (th-ffaeae). The assertion is on the recalled CONTENT reaching the model, +// which is the behavior the seam exists for. + +const recalledFact = "always add shows to the smoo-hub watchlist" + +// allContentSeen flattens everything the model was sent this turn — the surface a recalled memory +// must show up in. +func allContentSeen(t *testing.T, mock *core.MockLlmProvider) string { + t.Helper() + var sb strings.Builder + for _, call := range mock.Calls() { + blob, err := json.Marshal(call.Messages) + if err != nil { + t.Fatalf("marshal messages: %v", err) + } + sb.Write(blob) + } + return sb.String() +} + +func memoryWithEntry() core.Memory { + m := &core.InMemoryMemory{} + m.Remember(recalledFact) + return m +} + +// runMemoryTurn drives one plain text turn with the given provider installed the way the server +// installs it (post-construction, alongside hooks). +func runMemoryTurn(t *testing.T, provider MemoryProvider, message string) *core.MockLlmProvider { + t.Helper() + mock := core.NewMockLlmProvider().PushText("ok") + d := NewFrameDispatcher(NewInMemorySessionStore(), mock, AccessContext{}, "BASE PROMPT", nil, nil, nil, nil, nil, "", nil, nil, nil, nil) + d.memoryProvider = provider + sid := createSessionForTest(t, d) + + sink := sendAndWait(d, `{"action":"send_message","requestId":"r-1","sessionId":"`+sid+`","message":"`+message+`"}`) + if sink.find("eventual_response") == nil { + t.Fatal("turn did not complete") + } + return mock +} + +// ── rust: no_memory_means_no_recall_injection ──────────────────────────────── + +// Default: no provider ⇒ no auto-recall. Guards against the seam injecting when absent — an unopted +// deployment's turn must be byte-for-byte what it was before. +func TestNoMemoryMeansNoRecallInjection(t *testing.T) { + mock := runMemoryTurn(t, nil, "add shows to my watchlist") + if strings.Contains(allContentSeen(t, mock), recalledFact) { + t.Error("a turn with no memory provider carried a recalled memory") + } +} + +// A provider returning nil is the same as no provider — the seam must not fabricate a store just +// because one was installed. +func TestProviderReturningNilMeansNoRecallInjection(t *testing.T) { + mock := runMemoryTurn(t, NewStaticMemoryProvider(nil), "add shows to my watchlist") + if strings.Contains(allContentSeen(t, mock), recalledFact) { + t.Error("a declining provider still injected a recalled memory") + } +} + +// ── rust: attached_memory_is_auto_recalled_into_the_turn ───────────────────── + +// With a store attached the engine recalls the entries relevant to the user's message and injects +// them into the turn — the seam that lights up Big Smooth's durable auto-recall. +func TestAttachedMemoryIsAutoRecalledIntoTheTurn(t *testing.T) { + // The message shares "add", "shows", "watchlist" with the stored entry, so the engine's + // word-overlap recall surfaces it. + mock := runMemoryTurn(t, NewStaticMemoryProvider(memoryWithEntry()), "add shows to my watchlist") + if !strings.Contains(allContentSeen(t, mock), recalledFact) { + t.Errorf("an attached memory was not recalled into the turn: %s", allContentSeen(t, mock)) + } +} + +// An unrelated message recalls nothing: relevance-gated by the engine, not a blanket dump of every +// stored memory into every turn. The message shares NO token with the entry — the bundled lexical +// scorer counts raw token overlap with no stopword filter, so a single shared "the" would be enough +// to score a hit. +func TestIrrelevantMessageRecallsNothing(t *testing.T) { + mock := runMemoryTurn(t, NewStaticMemoryProvider(memoryWithEntry()), "explain quantum entanglement") + if strings.Contains(allContentSeen(t, mock), recalledFact) { + t.Error("an unrelated message recalled a memory it shares no token with") + } +} + +// The seam is access-scoped (mirroring knowledge) so a multi-tenant host can bind memory to the +// requester — the argument must actually reach the provider. +func TestProviderSeesTheCallersAccess(t *testing.T) { + rec := &recordingMemoryProvider{} + runMemoryTurn(t, rec, "hello") + if rec.calls != 1 { + t.Fatalf("provider called %d times, want exactly 1", rec.calls) + } +} + +type recordingMemoryProvider struct{ calls int } + +func (r *recordingMemoryProvider) MemoryForAccess(_ AccessContext) core.Memory { + r.calls++ + return nil +} diff --git a/go/server/server.go b/go/server/server.go index 86cdc229..2e58472d 100644 --- a/go/server/server.go +++ b/go/server/server.go @@ -32,6 +32,10 @@ type Server struct { // Nil → fall back to the env-configured DirSkillResolver (SMOOTH_SKILLS_DIR); with // neither, any skill field is a clean SKILL_NOT_FOUND. skills SkillResolver + // memory supplies each turn's durable-recall store, scoped to the connection's access + // (th-ebe27d / Rust #330). Nil → no auto-recall, which is every deployment that has not + // opted in. + memory MemoryProvider // tools are registered with the agent on every turn (default none → no behavior // change). The dispatcher threads them into the turn runner, which passes them // straight to the engine AgentOptions; the engine drives the tool loop and the @@ -127,6 +131,10 @@ func WithSystemPrompt(p string) Option { return func(srv *Server) { srv.systemP // and any skill field is a clean SKILL_NOT_FOUND. func WithSkillResolver(r SkillResolver) Option { return func(srv *Server) { srv.skills = r } } +// WithMemoryProvider installs the provider that supplies each turn's durable-recall store. Omitted +// → no auto-recall, which is every deployment that has not opted in. +func WithMemoryProvider(p MemoryProvider) Option { return func(srv *Server) { srv.memory = p } } + // WithTools registers the engine tools the agent may call during a turn (default none). // Threaded into every turn via the dispatcher → turn runner → engine AgentOptions. func WithTools(tools []core.Tool) Option { return func(srv *Server) { srv.tools = tools } } @@ -360,6 +368,7 @@ func (s *Server) connectionLoop(conn *websocket.Conn, access AccessContext) { // construction (before the dispatcher serves any frame) to avoid churning the // long constructor signature. Nil → no hooks. dispatcher.hooks = s.hooks + dispatcher.memoryProvider = s.memory // Same post-construction seam as hooks. Explicit resolver wins, else the // env-configured default, else off — mirrors the Rust/TS rule. if s.skills != nil { diff --git a/go/server/turn_runner.go b/go/server/turn_runner.go index fc9b0dc0..05b818c7 100644 --- a/go/server/turn_runner.go +++ b/go/server/turn_runner.go @@ -77,6 +77,10 @@ type TurnRunner struct { // around every tool the agent dispatches this turn (nil → none). Set by the // dispatcher after construction, alongside tools. hooks []core.ToolHook + // memory is the durable-recall store for this turn, already resolved for the connection's + // access by the dispatcher (th-ebe27d / Rust #330). Set after construction alongside hooks — + // the constructor signature is long enough. nil → the turn runs without auto-recall. + memory core.Memory // knowledge is the retriever (already SCOPED to the connection's access) the agent // grounds on. When set, the runner also queries it with the user message (top // autoContextLimit) to build the turn's auto-context citations — the sources the @@ -248,6 +252,10 @@ func (r *TurnRunner) Run(ctx context.Context, sessionID, conversationID, request // citations built above. r.systemPrompt is already assembled (base + per-agent // config + current workflow step) by the caller. opts := core.AgentOptions{Instructions: r.systemPrompt, Tools: r.tools, Knowledge: r.knowledge, Hooks: r.hooks} + // Durable auto-recall: with a store attached the engine pulls the entries relevant to the + // user's message into context. nil (every deployment that has not opted in) leaves the turn + // byte-for-byte unchanged. + opts.Memory = r.memory // Raised, no-longer-starving turn defaults (8192 / 20), with max_tokens clamped to // the model's output ceiling when known (nil ⇒ the raised default). Setting these // explicitly (rather than relying on the engine defaults) keeps the server robust diff --git a/python/server/src/smooth_operator_server/__init__.py b/python/server/src/smooth_operator_server/__init__.py index 6d798ae5..c805ee50 100644 --- a/python/server/src/smooth_operator_server/__init__.py +++ b/python/server/src/smooth_operator_server/__init__.py @@ -37,6 +37,7 @@ InteractionRequest, PendingInteractions, ) +from .memory import MemoryProvider, StaticMemoryProvider from .otp import ( OtpChannel, OtpContact, @@ -86,6 +87,8 @@ "coding_tools_from_env", "resolve_workspace_path", "FrameDispatcher", + "MemoryProvider", + "StaticMemoryProvider", "SKILLS_DIR_ENV", "DirSkillResolver", "SkillResolver", diff --git a/python/server/src/smooth_operator_server/dispatcher.py b/python/server/src/smooth_operator_server/dispatcher.py index b9990c08..c8ecec48 100644 --- a/python/server/src/smooth_operator_server/dispatcher.py +++ b/python/server/src/smooth_operator_server/dispatcher.py @@ -33,6 +33,7 @@ from .backplane import Target from .confirmation import ConfirmationRegistry from .interaction import InteractionOutcome, InteractionRegistry, PendingInteractions +from .memory import MemoryProvider from .otp import OtpContact, OtpInvalid, OtpService, OtpVerified from .session_store import SessionStore from .skills import SkillResolver, resolve_section @@ -66,6 +67,7 @@ def __init__( otp_service: OtpService | None = None, associate: Callable[[Target], Awaitable[None]] | None = None, skill_resolver: SkillResolver | None = None, + memory_provider: MemoryProvider | None = None, ) -> None: self._store = store self._chat_client = chat_client @@ -76,6 +78,9 @@ def __init__( #: None → the feature is off and any `skill` field is a clean SKILL_NOT_FOUND, #: so a multi-tenant deploy never serves host skills by accident. self._skill_resolver = skill_resolver + #: Supplies the turn's durable-recall store, scoped to this connection's access + #: (th-ebe27d / Rust #330). None → no auto-recall, the default. + self._memory_provider = memory_provider self._model = model self._tools = tools or [] #: Tool-name patterns gated behind human confirmation (empty → HITL off). @@ -586,6 +591,10 @@ async def _handle_send_message(self, frame: dict, request_id: str | None, sink: self._chat_client, self._store, knowledge=self._knowledge, + # Durable auto-recall, scoped the way retrieval is: the host decides which memory + # this caller recalls from. None (no provider, or one that declines this access) + # leaves the turn without auto-recall — byte-for-byte unchanged. + memory=(self._memory_provider.memory_for_access(self._access) if self._memory_provider else None), system_prompt=self._system_prompt, skill_section=skill_section_for_turn, model=self._model, diff --git a/python/server/src/smooth_operator_server/memory.py b/python/server/src/smooth_operator_server/memory.py new file mode 100644 index 00000000..872d96bd --- /dev/null +++ b/python/server/src/smooth_operator_server/memory.py @@ -0,0 +1,44 @@ +"""Durable auto-recall — the server side of the engine's ``Memory`` seam (Rust PR #330). + +The engine already knows how to auto-recall: give ``AgentOptions.memory`` a store and it +pulls the entries relevant to the user's message into the turn's context. What was missing +on this server is the way for a HOST to say *which* store — so every turn ran without +auto-recall regardless of what the deployment had. + +Mirrors the Rust ``StorageAdapter::memory_for_access`` seam. ``access`` is threaded (as it is +for knowledge) so a multi-tenant backend can bind memory to the requester's org/user; a +single-tenant host — Big Smooth's daemon, the reason this seam exists — ignores it and +returns its one store. +""" + +from __future__ import annotations + +from typing import Protocol, runtime_checkable + +from smooth_operator_core import Memory + +from .auth import AccessContext + + +@runtime_checkable +class MemoryProvider(Protocol): + """Supplies the durable-recall handle for a turn.""" + + def memory_for_access(self, access: AccessContext) -> Memory | None: + """The memory to auto-recall from for a caller with this access, or ``None`` for none. + + ``None`` is the default for every deployment that has not opted in, and leaves the + turn byte-for-byte unchanged.""" + ... + + +class StaticMemoryProvider: + """A :class:`MemoryProvider` over one unscoped store — the single-tenant case (Big Smooth's + daemon hands its SQLite-backed store straight through). A multi-tenant host implements the + protocol itself and keys off ``access`` instead.""" + + def __init__(self, memory: Memory | None) -> None: + self._memory = memory + + def memory_for_access(self, access: AccessContext) -> Memory | None: + return self._memory diff --git a/python/server/src/smooth_operator_server/server.py b/python/server/src/smooth_operator_server/server.py index 09b27d7d..aec98228 100644 --- a/python/server/src/smooth_operator_server/server.py +++ b/python/server/src/smooth_operator_server/server.py @@ -38,6 +38,7 @@ from .confirmation import ConfirmationRegistry from .dispatcher import FrameDispatcher from .interaction import InteractionRegistry, PendingInteractions +from .memory import MemoryProvider from .otp import OtpService from .session_store import InMemorySessionStore, SessionStore from .skills import DirSkillResolver, SkillResolver @@ -69,6 +70,10 @@ class ServerState: #: (``SMOOTH_SKILLS_DIR``); with neither, any ``skill`` field is a clean #: ``SKILL_NOT_FOUND``, so a multi-tenant deploy never serves host skills by accident. skill_resolver: SkillResolver | None = None + #: Supplies each turn's durable-recall store, scoped to the connection's access + #: (th-ebe27d / Rust #330). ``None`` → no auto-recall, which is every deployment + #: that has not opted in. + memory_provider: MemoryProvider | None = None model: str | None = None #: Tools the agent may call during a turn (default none). Each is an engine #: ``FunctionTool``/``Tool``; the turn runner passes them straight to the agent. @@ -172,6 +177,7 @@ async def associate(target: Target) -> None: # Mirrors the Rust/TS rule: explicit resolver wins, else the env-configured # default, else off. skill_resolver=state.skill_resolver or DirSkillResolver.from_env(), + memory_provider=state.memory_provider, model=state.model, tools=state.tools, confirm_tools=state.confirm_tools, diff --git a/python/server/src/smooth_operator_server/turn_runner.py b/python/server/src/smooth_operator_server/turn_runner.py index 4e6a98e0..eb61099b 100644 --- a/python/server/src/smooth_operator_server/turn_runner.py +++ b/python/server/src/smooth_operator_server/turn_runner.py @@ -29,6 +29,7 @@ HumanApprovalResponse, InProcessExecutor, Knowledge, + Memory, SmoothAgent, SmoothAgentThread, TextEvent, @@ -241,6 +242,7 @@ def __init__( chat_client: Any, store: SessionStore, knowledge: Knowledge | None = None, + memory: Memory | None = None, system_prompt: str | None = None, skill_section: str | None = None, model: str | None = None, @@ -265,6 +267,9 @@ def __init__( #: verbatim delegation to ``SmoothAgent.run_stream`` — behavior unchanged. self._executor = select_turn_executor(executor) self._knowledge = knowledge + #: Durable-recall store for this turn, already resolved for the connection's access + #: by the dispatcher (th-ebe27d / Rust #330). ``None`` → no auto-recall. + self._memory = memory self._system_prompt = system_prompt or DEFAULT_SYSTEM_PROMPT #: Pre-rendered `## Skill: ` section for THIS turn (th-ebe27d / Rust #: #338), already resolved by the dispatcher. Appended last in @@ -389,6 +394,11 @@ async def run( "max_tokens": DEFAULT_MAX_TOKENS, "max_iterations": DEFAULT_MAX_ITERATIONS, } + # Durable auto-recall: with a store attached the engine pulls the entries relevant to + # the user's message into context. None (every deployment that has not opted in) leaves + # the turn byte-for-byte unchanged. + if self._memory is not None: + options_kwargs["memory"] = self._memory if agent_tools: options_kwargs["tools"] = agent_tools if self._model is not None: diff --git a/python/server/tests/test_memory_provider.py b/python/server/tests/test_memory_provider.py new file mode 100644 index 00000000..3396e11b --- /dev/null +++ b/python/server/tests/test_memory_provider.py @@ -0,0 +1,116 @@ +"""Durable auto-recall parity (th-ebe27d / Rust PR #330 — the +``StorageAdapter::memory_for_access`` seam, tested in +``rust/smooth-operator-server/tests/injection_seams.rs``). + +The engine already knew how to recall; what was missing on this server was the host's way +to say WHICH store, so every turn ran without auto-recall no matter what the deployment had. +Tests are named after their Rust counterparts so a parity gap stays visible. + +The recall block's header text is deliberately NOT asserted: the five cores currently inject +three different strings for it (th-ffaeae). The assertion is on the recalled CONTENT reaching +the model, which is the behavior the seam exists for. +""" + +from __future__ import annotations + +import json + +import pytest +from smooth_operator_core import InMemoryMemory, MockLlmProvider + +from smooth_operator_server.agent_config import StaticAgentConfigResolver +from smooth_operator_server.auth import AccessContext +from smooth_operator_server.dispatcher import FrameDispatcher +from smooth_operator_server.memory import StaticMemoryProvider +from smooth_operator_server.session_store import InMemorySessionStore + +RECALLED = "always add shows to the smoo-hub watchlist" + + +def _all_content_seen(chat: MockLlmProvider) -> str: + """Everything the model was sent this turn, flattened — the surface a recalled memory + must show up in.""" + return json.dumps([c.messages for c in chat.calls]) + + +async def _run_turn(provider, message: str) -> MockLlmProvider: + store = InMemorySessionStore() + session = await store.create_session("agent-x", None, None) + chat = MockLlmProvider() + chat.push_text("ok") + dispatcher = FrameDispatcher( + store, + chat, + tools=[], + agent_config_resolver=StaticAgentConfigResolver({}), + memory_provider=provider, + ) + await dispatcher.dispatch( + json.dumps({"action": "send_message", "sessionId": session.session_id, "message": message}), + lambda _e: None, + ) + await dispatcher.wait_for_turns() + return chat + + +def _memory_with_entry() -> InMemoryMemory: + memory = InMemoryMemory() + memory.remember(RECALLED) + return memory + + +# ── rust: no_memory_means_no_recall_injection ──────────────────────────────── + + +@pytest.mark.asyncio +async def test_no_memory_means_no_recall_injection() -> None: + """Default: no provider ⇒ no auto-recall. Guards against the seam injecting when absent — + an unopted deployment's turn must be byte-for-byte what it was before.""" + chat = await _run_turn(None, "add shows to my watchlist") + assert RECALLED not in _all_content_seen(chat) + + +@pytest.mark.asyncio +async def test_provider_returning_none_means_no_recall_injection() -> None: + """A provider that returns None for this caller is the same as no provider — the seam must + not fabricate a store just because one was installed.""" + chat = await _run_turn(StaticMemoryProvider(None), "add shows to my watchlist") + assert RECALLED not in _all_content_seen(chat) + + +# ── rust: attached_memory_is_auto_recalled_into_the_turn ───────────────────── + + +@pytest.mark.asyncio +async def test_attached_memory_is_auto_recalled_into_the_turn() -> None: + """With a store attached the engine recalls the entries relevant to the user's message and + injects them into the turn — the seam that lights up Big Smooth's durable auto-recall.""" + # The message shares "add", "shows", "watchlist" with the stored entry, so the engine's + # word-overlap recall surfaces it. + chat = await _run_turn(StaticMemoryProvider(_memory_with_entry()), "add shows to my watchlist") + assert RECALLED in _all_content_seen(chat) + + +@pytest.mark.asyncio +async def test_irrelevant_message_recalls_nothing() -> None: + """An unrelated message recalls nothing: the seam is relevance-gated by the engine, not a + blanket dump of every stored memory into every turn. The message shares NO token with the + entry — the bundled lexical scorer counts raw token overlap with no stopword filter, so a + single shared "the" would be enough to score a hit.""" + chat = await _run_turn(StaticMemoryProvider(_memory_with_entry()), "explain quantum entanglement") + assert RECALLED not in _all_content_seen(chat) + + +@pytest.mark.asyncio +async def test_provider_sees_the_callers_access() -> None: + """The seam is access-scoped (mirroring knowledge) so a multi-tenant host can bind memory to + the requester — the argument must actually reach the provider.""" + seen: list[AccessContext] = [] + + class _Recording: + def memory_for_access(self, access: AccessContext): + seen.append(access) + return None + + await _run_turn(_Recording(), "hello") + assert len(seen) == 1 diff --git a/python/uv.lock b/python/uv.lock index a480d6c6..f74a277f 100644 --- a/python/uv.lock +++ b/python/uv.lock @@ -775,7 +775,7 @@ wheels = [ [[package]] name = "smooai-smooth-operator" -version = "1.51.0" +version = "1.58.7" source = { editable = "." } dependencies = [ { name = "jsonschema" }, diff --git a/typescript/server/src/frameDispatcher.ts b/typescript/server/src/frameDispatcher.ts index 71404a79..42725b7d 100644 --- a/typescript/server/src/frameDispatcher.ts +++ b/typescript/server/src/frameDispatcher.ts @@ -16,6 +16,7 @@ import type { StoredSession } from './sessionStore.js'; import { randomUUID } from 'node:crypto'; import { type AgentConfigResolver, assembleSystemPrompt } from './agentConfig.js'; +import type { MemoryProvider } from './memory.js'; import { resolveSection, type SkillResolver } from './skills.js'; import { gateTools, type SessionAuthenticator } from './toolGating.js'; import { ANONYMOUS_ACCESS, type AccessContext } from './auth.js'; @@ -53,6 +54,8 @@ export interface FrameDispatcherOptions { * `SKILL_NOT_FOUND`, so a multi-tenant deploy never serves host skills by accident. */ skillResolver?: SkillResolver; + /** Supplies the turn's durable-recall store, scoped to this connection's access (Rust #330). */ + memoryProvider?: MemoryProvider; /** Tools the agent may call during a turn (default none); forwarded to the {@link TurnRunner}. */ tools?: Tool[]; /** @@ -136,6 +139,7 @@ export class FrameDispatcher { associate?: (target: Target) => void; private readonly systemPrompt?: string; private readonly skillResolver?: SkillResolver; + private readonly memoryProvider?: MemoryProvider; private readonly tools: Tool[]; private readonly toolHooks: ToolHook[]; private readonly confirmTools: string[]; @@ -168,6 +172,7 @@ export class FrameDispatcher { this.access = options.access ?? ANONYMOUS_ACCESS; this.systemPrompt = options.systemPrompt; this.skillResolver = options.skillResolver; + this.memoryProvider = options.memoryProvider; this.tools = options.tools ?? []; this.toolHooks = options.toolHooks ?? []; this.confirmTools = options.confirmTools ?? []; @@ -723,6 +728,10 @@ export class FrameDispatcher { chatClient: this.chatClient, store: this.store, knowledge: scopedKnowledge, + // Durable auto-recall, scoped the way retrieval is: the host decides which memory this + // caller recalls from. Absent (no provider, or one that declines this access) leaves the + // turn without auto-recall — byte-for-byte unchanged. + memory: this.memoryProvider?.memoryForAccess(this.access), systemPrompt: effectiveSystemPrompt, tools: effectiveTools, toolHooks: this.toolHooks, diff --git a/typescript/server/src/index.ts b/typescript/server/src/index.ts index 5be5ee89..fbee4dce 100644 --- a/typescript/server/src/index.ts +++ b/typescript/server/src/index.ts @@ -37,6 +37,7 @@ export type { IntakeField, IntakeFieldError, IntakeFieldKey, IntakeValues } from export { DEFAULT_MAX_ITERATIONS, DEFAULT_MAX_TOKENS, DEFAULT_MODEL, DEFAULT_SYSTEM_PROMPT, TurnRunner } from './turnRunner.js'; export type { Sink, TurnResult, TurnRunnerOptions } from './turnRunner.js'; +export { StaticMemoryProvider, type MemoryProvider } from './memory.js'; export { DirSkillResolver, isValidSkillName, resolveSection, skillSection, stripFrontmatter, SKILLS_DIR_ENV, type SkillResolver } from './skills.js'; export { assembleSystemPrompt, parseAgentConfig, StaticAgentConfigResolver } from './agentConfig.js'; export type { AgentConfig, AgentConfigResolver, EnabledTool } from './agentConfig.js'; diff --git a/typescript/server/src/memory.ts b/typescript/server/src/memory.ts new file mode 100644 index 00000000..c9df3e4b --- /dev/null +++ b/typescript/server/src/memory.ts @@ -0,0 +1,38 @@ +/** + * Durable auto-recall — the server side of the engine's `Memory` seam (Rust PR #330). + * + * The engine already knows how to auto-recall: give `AgentOptions.memory` a store and it pulls the + * entries relevant to the user's message into the turn's context. What was missing on this server is + * the way for a HOST to say *which* store — so every turn ran without auto-recall regardless of what + * the deployment had. + * + * Mirrors the Rust `StorageAdapter::memory_for_access` seam. `access` is threaded (as it is for + * knowledge) so a multi-tenant backend can bind memory to the requester's org/user; a single-tenant + * host — Big Smooth's daemon, the reason this seam exists — ignores it and returns its one store. + */ +import type { Memory } from '@smooai/smooth-operator-core'; + +import type { AccessContext } from './auth.js'; + +/** Supplies the durable-recall handle for a turn. */ +export interface MemoryProvider { + /** + * The memory to auto-recall from for a caller with this access, or `undefined` for none. + * `undefined` is the default for every deployment that has not opted in, and leaves the turn + * byte-for-byte unchanged. + */ + memoryForAccess(access: AccessContext): Memory | undefined; +} + +/** + * A {@link MemoryProvider} over one unscoped store — the single-tenant case (Big Smooth's daemon + * hands its SQLite-backed store straight through). A multi-tenant host implements the interface + * itself and keys off `access` instead. + */ +export class StaticMemoryProvider implements MemoryProvider { + constructor(private readonly memory: Memory | undefined) {} + + memoryForAccess(_access: AccessContext): Memory | undefined { + return this.memory; + } +} diff --git a/typescript/server/src/server.ts b/typescript/server/src/server.ts index 3e8c1f8d..1e19e4b4 100644 --- a/typescript/server/src/server.ts +++ b/typescript/server/src/server.ts @@ -27,6 +27,7 @@ import type { AgentConfigResolver } from './agentConfig.js'; import type { SessionAuthenticator } from './toolGating.js'; import type { OtpService } from './otp.js'; import { type AccessKnowledge, FrameDispatcher } from './frameDispatcher.js'; +import type { MemoryProvider } from './memory.js'; import { DirSkillResolver, type SkillResolver } from './skills.js'; import type { ModelCeilingResolver } from './modelCeiling.js'; import type { ToolContext, ToolProvider } from './toolContext.js'; @@ -120,6 +121,12 @@ export interface ServerOptions { * `SKILL_NOT_FOUND`, so a multi-tenant deploy never serves host skills by accident. */ skillResolver?: SkillResolver; + + /** + * Supplies each turn's durable-recall store, scoped to the connection's access (th-ebe27d / + * Rust #330). Omitted ⇒ no auto-recall, which is every deployment that has not opted in. + */ + memoryProvider?: MemoryProvider; /** * The Rich Interactions the server hosts (see `interaction.ts`). Each turn registers * one `request_` raise tool per kind, gated per-kind by the session's declared @@ -206,6 +213,7 @@ export function buildServer(options: ServerOptions): { toolProvider: options.toolProvider, // Mirrors Rust's install_skill_resolver_from_env: explicit wins, else env, else off. skillResolver: options.skillResolver ?? DirSkillResolver.fromEnv(), + memoryProvider: options.memoryProvider, }); // Fire-and-forget the per-connection loop; it owns the socket's lifecycle. void runConnection(socket, dispatcher, backplane, drain.signal); diff --git a/typescript/server/src/turnRunner.ts b/typescript/server/src/turnRunner.ts index 5dd22f49..1ea6af2f 100644 --- a/typescript/server/src/turnRunner.ts +++ b/typescript/server/src/turnRunner.ts @@ -12,7 +12,7 @@ * `runStream` mapped event-by-event onto protocol events. */ import { approve, deny, SmoothAgent } from '@smooai/smooth-operator-core'; -import type { AgentExecutor, AgentOptions, ChatClientLike, HumanApprovalRequest, HumanApprovalResponse, Knowledge, StreamEvent, Tool, ToolHook } from '@smooai/smooth-operator-core'; +import type { AgentExecutor, AgentOptions, ChatClientLike, HumanApprovalRequest, HumanApprovalResponse, Knowledge, Memory, StreamEvent, Tool, ToolHook } from '@smooai/smooth-operator-core'; import type { ConfirmationRegistry } from './confirmation.js'; import { turnExecutor } from './executorSelection.js'; @@ -166,6 +166,11 @@ export interface TurnRunnerOptions { store: SessionStore; /** Optional knowledge retriever, already SCOPED to the connection's access (ACL). */ knowledge?: Knowledge; + /** + * Durable-recall store for this turn, already resolved for the connection's access by the + * dispatcher (th-ebe27d / Rust #330). Absent → the turn runs without auto-recall. + */ + memory?: Memory; systemPrompt?: string; /** Tools the agent may call during the turn (default none); passed straight to the engine. */ tools?: Tool[]; @@ -245,6 +250,7 @@ export class TurnRunner { private readonly chatClient: ChatClientLike; private readonly store: SessionStore; private readonly knowledge?: Knowledge; + private readonly memory?: Memory; private readonly systemPrompt: string; private readonly tools: Tool[]; private readonly toolHooks: ToolHook[]; @@ -271,6 +277,7 @@ export class TurnRunner { this.chatClient = options.chatClient; this.store = options.store; this.knowledge = options.knowledge; + this.memory = options.memory; this.systemPrompt = options.systemPrompt ?? DEFAULT_SYSTEM_PROMPT; this.tools = options.tools ?? []; this.toolHooks = options.toolHooks ?? []; @@ -346,6 +353,10 @@ export class TurnRunner { maxIterations: DEFAULT_MAX_ITERATIONS, }; if (this.knowledge) agentOptions.knowledge = this.knowledge; + // Durable auto-recall: with a store attached the engine pulls the entries relevant to the + // user's message into context. Absent (every deployment that has not opted in) leaves the + // turn byte-for-byte unchanged. + if (this.memory) agentOptions.memory = this.memory; if (this.tools.length > 0) agentOptions.tools = this.tools; // Thread consumer-supplied surveillance hooks into the engine's per-turn tool // registry. Empty ⇒ unset ⇒ behaviour unchanged. diff --git a/typescript/server/test/memoryProvider.test.ts b/typescript/server/test/memoryProvider.test.ts new file mode 100644 index 00000000..5a4d7ece --- /dev/null +++ b/typescript/server/test/memoryProvider.test.ts @@ -0,0 +1,100 @@ +/** + * Durable auto-recall (Rust PR #330 — the `StorageAdapter::memory_for_access` seam, tested in + * `rust/smooth-operator-server/tests/injection_seams.rs`) — the TS server's parity. + * + * The engine already knew how to recall; what was missing on this server was the host's way to say + * WHICH store, so every turn ran without auto-recall no matter what the deployment had. Tests are + * named after their Rust counterparts so a parity gap stays visible. + * + * The recall block's header text is deliberately NOT asserted: the five cores currently inject three + * different strings for it (th-ffaeae). The assertion is on the recalled CONTENT reaching the model, + * which is the behavior the seam exists for. + */ +import { InMemoryMemory, MockLlmProvider } from '@smooai/smooth-operator-core'; +import { afterEach, describe, expect, it } from 'vitest'; + +import type { AccessContext } from '../src/auth.js'; +import { StaticMemoryProvider, type MemoryProvider } from '../src/memory.js'; +import { serve, type RunningServer } from '../src/server.js'; +import { TestClient } from './wsClient.js'; + +const RECALLED = 'always add shows to the smoo-hub watchlist'; + +/** Everything the model was sent this turn, flattened — the surface a recalled memory must show up in. */ +function allContentSeen(chat: MockLlmProvider): string { + return JSON.stringify(chat.calls.map((c) => c.messages)); +} + +function memoryWithEntry(): InMemoryMemory { + const memory = new InMemoryMemory(); + memory.remember(RECALLED); + return memory; +} + +describe('durable auto-recall (over the wire)', () => { + let server: RunningServer | undefined; + afterEach(async () => { + await server?.close(); + server = undefined; + }); + + async function runTurn(memoryProvider: MemoryProvider | undefined, message: string): Promise { + const chat = new MockLlmProvider().pushText('ok'); + server = await serve({ chatClient: chat, memoryProvider }); + const client = await TestClient.connect(server.url); + client.sendAction({ action: 'create_conversation_session', requestId: 'cs', agentId: 'agent' }); + const sessionId = ((await client.receive()).data as Record).sessionId as string; + + client.sendAction({ action: 'send_message', requestId: 'r2', sessionId, message }); + await client.receiveUntil('eventual_response'); + await client.close(); + return chat; + } + + // ── rust: no_memory_means_no_recall_injection ──────────────────────────── + + it('injects no recall when no provider is installed', async () => { + // Guards against the seam injecting when absent — an unopted deployment's turn must be + // byte-for-byte what it was before. + const chat = await runTurn(undefined, 'add shows to my watchlist'); + expect(allContentSeen(chat)).not.toContain(RECALLED); + }); + + it('injects no recall when the provider declines this caller', async () => { + // A provider returning undefined is the same as no provider — the seam must not fabricate a + // store just because one was installed. + const chat = await runTurn(new StaticMemoryProvider(undefined), 'add shows to my watchlist'); + expect(allContentSeen(chat)).not.toContain(RECALLED); + }); + + // ── rust: attached_memory_is_auto_recalled_into_the_turn ───────────────── + + it('auto-recalls an attached memory into the turn', async () => { + // The message shares "add", "shows", "watchlist" with the stored entry, so the engine's + // word-overlap recall surfaces it. + const chat = await runTurn(new StaticMemoryProvider(memoryWithEntry()), 'add shows to my watchlist'); + expect(allContentSeen(chat)).toContain(RECALLED); + }); + + it('recalls nothing for an unrelated message', async () => { + // Relevance-gated by the engine, not a blanket dump of every stored memory into every turn. + // The message shares NO token with the entry — the bundled lexical scorer counts raw token + // overlap with no stopword filter, so a single shared "the" would be enough to score a hit. + const chat = await runTurn(new StaticMemoryProvider(memoryWithEntry()), 'explain quantum entanglement'); + expect(allContentSeen(chat)).not.toContain(RECALLED); + }); + + it('hands the provider the caller access, so a multi-tenant host can scope by requester', async () => { + const seen: AccessContext[] = []; + await runTurn( + { + memoryForAccess(access: AccessContext) { + seen.push(access); + return undefined; + }, + }, + 'hello', + ); + expect(seen).toHaveLength(1); + }); +});