From f4a10a4f50a44113000f8691d03872bb473819bd Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Wed, 7 Oct 2026 07:58:10 +0300 Subject: [PATCH 1/3] refactor(cache): move response cache and prompt layout to host harness The cache module now only carries the per-request CachePolicy and the canonical_value JSON helper, since the response cache, its keys and the prompt-layout guard live in the host harness. canonical_value is exported so host-derived cache keys canonicalize identically, and the sha2 dependency is dropped along with the removed hashing code. Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinyinference-llm/Cargo.toml | 1 - crates/tinyinference-llm/src/cache/mod.rs | 345 +++--------------- .../tinyinference-llm/src/cache/mod_tests.rs | 260 ++----------- crates/tinyinference-llm/src/cache/types.rs | 139 ------- 4 files changed, 81 insertions(+), 664 deletions(-) delete mode 100644 crates/tinyinference-llm/src/cache/types.rs diff --git a/crates/tinyinference-llm/Cargo.toml b/crates/tinyinference-llm/Cargo.toml index 21cfa895..7b1a0c8b 100644 --- a/crates/tinyinference-llm/Cargo.toml +++ b/crates/tinyinference-llm/Cargo.toml @@ -19,7 +19,6 @@ futures = { workspace = true } reqwest = { workspace = true } serde = { workspace = true } serde_json = { workspace = true } -sha2 = { workspace = true } thiserror = { workspace = true } tinyinference-core = { version = "0.3.0", path = "../tinyinference-core" } # The model-facing tool protocol: prompt-guided rendering, text-mode parsing, diff --git a/crates/tinyinference-llm/src/cache/mod.rs b/crates/tinyinference-llm/src/cache/mod.rs index 70d30de0..9e26326a 100644 --- a/crates/tinyinference-llm/src/cache/mod.rs +++ b/crates/tinyinference-llm/src/cache/mod.rs @@ -1,300 +1,39 @@ -//! Harness cache module — prompt, response, and layout caches. +//! Per-request cache policy and JSON canonicalization. //! -//! In the recursive runtime the same request can recur many times — a -//! sub-agent re-asked an identical sub-question, a graph node replayed during -//! recovery, or a deterministic test driving the loop twice. This module makes -//! that recursion cheap and deterministic: the local response cache short- -//! circuits an identical model call entirely (a consuming agent loop can emit -//! its own cache hit/miss events), while the prompt-cache -//! layout tooling protects the stable prefix the *provider* itself caches. +//! [`CachePolicy`] is the per-request knob a [`crate::model::ModelRequest`] +//! carries (`cache_policy`): whether a host may serve the call from its local +//! response cache, whether the provider prompt-cache prefix must be protected, +//! and the entry TTL and key namespace. This crate only carries the policy and +//! honours `protect_prompt_prefix` in the providers that map it onto wire +//! cache markers; the response cache itself, its keys and the prompt-layout +//! guard live in the host harness (`tinyagents-harness`'s `cache` module). //! -//! # Two distinct caching concerns -//! -//! ## 1. Local response cache -//! [`ResponseCache`] + [`InMemoryResponseCache`] let the harness skip provider -//! API calls entirely when it has already seen an identical request. Use -//! [`cache_key`] to produce a stable, deterministic key from a -//! [`crate::model::ModelRequest`]. -//! -//! ## 2. Provider prompt / KV-cache layout protection -//! [`PromptCacheLayout`] records the ordered cacheable prefix of a request. -//! [`CacheLayoutEvent`] describes mutations so middleware can signal whether it -//! preserved or invalidated the provider's KV-cache prefix. -//! [`CachePolicy`] toggles both concerns at the call-site level. -//! -mod types; +//! [`canonical_value`] sorts JSON object keys so a host's cache keys hash +//! equal requests equally regardless of field insertion order. -use async_trait::async_trait; +use serde::{Deserialize, Serialize}; use serde_json::Value; -use sha2::{Digest, Sha256}; - -pub use types::*; - -use crate::model::{ModelRequest, ModelResponse}; -use crate::{Error, Result}; - -// ── Deterministic hash ──────────────────────────────────────────────────────── - -/// Renders a finalized SHA-256 digest as a 64-character lowercase hex string. -fn hex_digest(digest: impl AsRef<[u8]>) -> String { - digest - .as_ref() - .iter() - .map(|byte| format!("{byte:02x}")) - .collect() -} - -/// Folds one JSON `value` into `hasher` as a self-delimiting frame: an ASCII -/// domain `tag`, then the canonical byte length (little-endian `u64`), then the -/// canonical bytes. -/// -/// Canonicalizing per component keeps peak memory bounded by the single largest -/// value rather than the whole request tree, and the length prefix makes the -/// concatenation of frames unambiguous — no two distinct component sequences -/// can hash to the same byte stream. -fn fold_canonical(hasher: &mut Sha256, tag: u8, value: Value) { - let bytes = serde_json::to_vec(&canonical_value(value)) - .expect("serializing a serde_json::Value cannot fail"); - hasher.update([tag]); - hasher.update((bytes.len() as u64).to_le_bytes()); - hasher.update(&bytes); -} - -/// Computes a deterministic FNV-1a 64-bit hash over `data` and returns it as -/// a 16-character lowercase hex string. -/// -/// FNV-1a uses a fixed, seed-free offset basis so the result is identical -/// across process restarts — unlike Rust's default `SipHash`, which is seeded -/// randomly at startup. It is used only for short local prompt-layout -/// fingerprints, not for response-cache identity. -fn fnv1a_hex(data: &[u8]) -> String { - const OFFSET_BASIS: u64 = 14_695_981_039_346_656_037; - const PRIME: u64 = 1_099_511_628_211; - let mut hash = OFFSET_BASIS; - for &byte in data { - hash ^= u64::from(byte); - hash = hash.wrapping_mul(PRIME); - } - format!("{hash:016x}") -} - -/// Recursively sorts the keys of every JSON object so that the serialized form -/// is canonical regardless of insertion order. -fn canonical_value(v: Value) -> Value { - match v { - Value::Object(map) => { - let mut pairs: Vec<(String, Value)> = map.into_iter().collect(); - pairs.sort_by(|a, b| a.0.cmp(&b.0)); - Value::Object( - pairs - .into_iter() - .map(|(k, val)| (k, canonical_value(val))) - .collect(), - ) - } - Value::Array(arr) => Value::Array(arr.into_iter().map(canonical_value).collect()), - other => other, - } -} -// ── cache_key ───────────────────────────────────────────────────────────────── - -/// Produces a stable, deterministic cache key for `request`. -/// -/// The key is a 64-character lowercase SHA-256 hex string built by folding the -/// request into the hasher **incrementally**, one component at a time: -/// 1. Serialize `request` once to a [`serde_json::Value`]. -/// 2. Fold each conversation message as its own length-prefixed, canonicalized -/// frame (tag `M`), preceded by the message count. -/// 3. Fold each tool schema likewise (tag `T`), preceded by the tool count. -/// 4. Fold the remaining scalar/parameter fields — everything left in the -/// request object once `messages` and `tools` are removed — as one envelope -/// frame (tag `E`). -/// -/// This avoids the previous approach's three simultaneous whole-transcript -/// allocations (a full `Value` tree, a full canonical rebuild, and a full byte -/// buffer). Transcripts routinely carry large tool results; canonicalizing and -/// serializing per component bounds peak memory by the single largest component -/// instead of the entire request. The envelope is taken as "whatever remains" -/// so that any future [`ModelRequest`] field automatically participates in the -/// key — no field can silently drop out and cause a false cache hit. +/// Policy knobs controlling both response caching and provider prompt-cache +/// layout protection. /// -/// Distinct requests still map to distinct keys (modulo SHA-256 collision -/// resistance): per-component length prefixes make the frame stream -/// unambiguous, and every behavior-affecting field is folded exactly once. -/// -/// # Panics -/// Does not panic. If serialization unexpectedly fails, the affected frame -/// folds empty bytes; the key stays well-defined. -pub fn cache_key(request: &ModelRequest) -> String { - let mut hasher = Sha256::new(); - let mut root = serde_json::to_value(request).unwrap_or(Value::Null); - - if let Value::Object(map) = &mut root { - // Messages: fold one at a time so a long transcript never materializes - // a second full tree. The count frame keeps `[a, b]` distinct from a - // single message that happens to serialize to the same concatenation. - if let Some(Value::Array(messages)) = map.remove("messages") { - hasher.update(b"M"); - hasher.update((messages.len() as u64).to_le_bytes()); - for message in messages { - fold_canonical(&mut hasher, b'm', message); - } - } - // Tool schemas: already name-sorted by `ToolRegistry::schemas`, so the - // order is deterministic across calls. - if let Some(Value::Array(tools)) = map.remove("tools") { - hasher.update(b"T"); - hasher.update((tools.len() as u64).to_le_bytes()); - for tool in tools { - fold_canonical(&mut hasher, b't', tool); - } - } - } - - // Envelope: every remaining scalar/parameter field in one frame. - fold_canonical(&mut hasher, b'E', root); - hex_digest(hasher.finalize()) -} - -// ── InMemoryResponseCache ───────────────────────────────────────────────────── - -impl InMemoryResponseCache { - /// Default LRU capacity when constructed via [`new`](Self::new) or - /// [`Default`]. - pub const DEFAULT_CAPACITY: usize = 1024; - - /// Creates a new, empty in-memory response cache bounded by - /// [`DEFAULT_CAPACITY`](Self::DEFAULT_CAPACITY) entries. - pub fn new() -> Self { - Self::with_capacity(Self::DEFAULT_CAPACITY) - } - - /// Creates a new, empty in-memory response cache retaining at most - /// `capacity` entries (least-recently-used evicted first). A `capacity` of - /// zero is treated as `1` so the cache always retains the last write. - pub fn with_capacity(capacity: usize) -> Self { - Self { - inner: std::sync::Arc::new(std::sync::Mutex::new(LruResponseMap { - data: std::collections::HashMap::new(), - order: std::collections::VecDeque::new(), - capacity: capacity.max(1), - })), - } - } -} - -impl Default for InMemoryResponseCache { - fn default() -> Self { - Self::new() - } -} - -impl LruResponseMap { - /// Moves `key` to the most-recently-used end of the order queue. - fn touch(&mut self, key: &str) { - if let Some(pos) = self.order.iter().position(|k| k == key) { - let k = self.order.remove(pos).expect("position is valid"); - self.order.push_back(k); - } - } -} - -#[async_trait] -impl ResponseCache for InMemoryResponseCache { - async fn get(&self, key: &str) -> Result> { - let mut inner = self - .inner - .lock() - .map_err(|e| Error::Validation(format!("cache lock poisoned: {e}")))?; - let hit = inner.data.get(key).cloned(); - if hit.is_some() { - inner.touch(key); - } - Ok(hit) - } - - async fn put(&self, key: &str, value: ModelResponse) -> Result<()> { - let mut inner = self - .inner - .lock() - .map_err(|e| Error::Validation(format!("cache lock poisoned: {e}")))?; - if inner.data.insert(key.to_string(), value).is_some() { - // Existing key: refresh its recency without changing the length. - inner.touch(key); - } else { - inner.order.push_back(key.to_string()); - // Evict least-recently-used entries until within capacity. - while inner.order.len() > inner.capacity { - if let Some(evicted) = inner.order.pop_front() { - inner.data.remove(&evicted); - } - } - } - Ok(()) - } -} - -// ── PromptCacheLayout ───────────────────────────────────────────────────────── - -impl PromptCacheLayout { - /// Builds a [`PromptCacheLayout`] from `request` by collecting the ids of - /// all cacheable (stable) segments in their declared order. - /// - /// The fingerprint is a deterministic FNV-1a hash of the joined prefix ids - /// so regression tests can assert prefix stability independently of the - /// full request hash. - pub fn from_request(request: &ModelRequest) -> Self { - let prefix_ids: Vec = request.cacheable_prefix_ids(); - let fingerprint = fnv1a_hex( - format!( - "{}\0{}", - prefix_ids.join(","), - request.prompt_fingerprint.as_deref().unwrap_or_default() - ) - .as_bytes(), - ); - Self { - prefix_ids, - fingerprint, - content_fingerprint: request.prompt_fingerprint.clone(), - } - } - - /// Returns the ordered ids of cacheable (stable) prefix segments. - pub fn prefix_ids(&self) -> &[String] { - &self.prefix_ids - } - - /// Returns the deterministic fingerprint of the ordered prefix ids. - /// - /// Two layouts with identical `prefix_ids` produce the same fingerprint. - pub fn fingerprint(&self) -> &str { - &self.fingerprint - } - - /// Returns `true` if `self` and `other` have the same cacheable prefix ids - /// in the same order, meaning the provider KV-cache prefix is stable - /// across the two requests. - pub fn is_prefix_stable_against(&self, other: &PromptCacheLayout) -> bool { - self.prefix_ids == other.prefix_ids && self.content_fingerprint == other.content_fingerprint - } -} - -// ── CacheLayoutEvent ────────────────────────────────────────────────────────── - -impl CacheLayoutEvent { - /// Constructs a [`CacheLayoutEvent`] by comparing `before` and `after` - /// layouts, filling in the computed `changed_prefix` and `volatile_only` - /// flags automatically. - pub fn new(before: &PromptCacheLayout, after: &PromptCacheLayout) -> Self { - Self { - changed_prefix: !before.is_prefix_stable_against(after), - volatile_only: after.prefix_ids().is_empty(), - segment_ids_before: before.prefix_ids().to_vec(), - segment_ids_after: after.prefix_ids().to_vec(), - } - } +/// Both flags default to `false` (no caching / no protection) so a host is +/// safe-by-default and opts must be explicit. +#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct CachePolicy { + /// When `true`, the host may look up (and write) local response cache + /// entries before calling the provider. + pub response_cache_enabled: bool, + /// When `true`, middleware must preserve the order and content of cacheable + /// prefix segments, and providers that support it mark the stable prefix + /// for their own prompt cache. + pub protect_prompt_prefix: bool, + /// Entry time-to-live in milliseconds; `None` means no expiry. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub ttl_ms: Option, + /// Optional cache-key namespace. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub namespace: Option, } impl CachePolicy { @@ -324,6 +63,30 @@ impl CachePolicy { } } +/// Recursively sorts the keys of every JSON object so that the serialized form +/// is canonical regardless of insertion order. +/// +/// Public so every cache key derived from JSON — this crate's and a host's — +/// canonicalizes the same way; two copies that drifted would make equal +/// requests hash apart. +#[must_use] +pub fn canonical_value(v: Value) -> Value { + match v { + Value::Object(map) => { + let mut pairs: Vec<(String, Value)> = map.into_iter().collect(); + pairs.sort_by(|a, b| a.0.cmp(&b.0)); + Value::Object( + pairs + .into_iter() + .map(|(k, val)| (k, canonical_value(val))) + .collect(), + ) + } + Value::Array(arr) => Value::Array(arr.into_iter().map(canonical_value).collect()), + other => other, + } +} + #[cfg(test)] #[path = "mod_tests.rs"] mod test; diff --git a/crates/tinyinference-llm/src/cache/mod_tests.rs b/crates/tinyinference-llm/src/cache/mod_tests.rs index fd2d75d5..56bb9635 100644 --- a/crates/tinyinference-llm/src/cache/mod_tests.rs +++ b/crates/tinyinference-llm/src/cache/mod_tests.rs @@ -1,249 +1,43 @@ -//! Tests added in a later pass. -//! -//! Smoke tests confirming that [`super::InMemoryResponseCache`] round-trips a -//! response, that [`super::cache_key`] is deterministic, and that -//! [`super::PromptCacheLayout`] correctly detects stable vs. changed prefixes. +//! Tests for [`CachePolicy`] and [`canonical_value`]. use super::*; -use crate::message::Message; -use crate::model::{ModelRequest, ModelResponse, PromptSegment, SegmentRole}; -use crate::tool::ToolSchema; use serde_json::json; - -#[tokio::test] -async fn response_cache_put_get() { - let cache = InMemoryResponseCache::new(); - assert!(cache.get("k").await.unwrap().is_none()); - - let response = ModelResponse::assistant("hello"); - cache.put("k", response.clone()).await.unwrap(); - let fetched = cache.get("k").await.unwrap().expect("should be cached"); - assert_eq!(fetched.text(), "hello"); -} - -#[tokio::test] -async fn response_cache_evicts_least_recently_used() { - let cache = InMemoryResponseCache::with_capacity(2); - cache.put("a", ModelResponse::assistant("a")).await.unwrap(); - cache.put("b", ModelResponse::assistant("b")).await.unwrap(); - - // Touch "a" so "b" becomes the least-recently-used entry. - assert!(cache.get("a").await.unwrap().is_some()); - - // Inserting a third key evicts "b", not the recently-read "a". - cache.put("c", ModelResponse::assistant("c")).await.unwrap(); - assert!( - cache.get("a").await.unwrap().is_some(), - "a was recently used" - ); - assert!(cache.get("c").await.unwrap().is_some(), "c is newest"); - assert!(cache.get("b").await.unwrap().is_none(), "b evicted as LRU"); -} - -#[tokio::test] -async fn response_cache_capacity_zero_retains_last() { - // A zero capacity is clamped to 1: the cache still retains the last write. - let cache = InMemoryResponseCache::with_capacity(0); - cache.put("x", ModelResponse::assistant("x")).await.unwrap(); - cache.put("y", ModelResponse::assistant("y")).await.unwrap(); - assert!(cache.get("x").await.unwrap().is_none()); - assert!(cache.get("y").await.unwrap().is_some()); -} - -#[test] -fn cache_key_is_deterministic() { - let req = ModelRequest::new(vec![]).with_model("gpt-4"); - let k1 = cache_key(&req); - let k2 = cache_key(&req); - assert_eq!(k1, k2, "cache_key must be deterministic"); - assert_eq!(k1.len(), 64, "cache_key should be a SHA-256 hex digest"); - assert!(k1.chars().all(|c| c.is_ascii_hexdigit())); -} - -#[test] -fn prompt_layout_detects_changed_content_with_stable_ids() { - let segment = PromptSegment { - id: "system".into(), - role: SegmentRole::System, - cacheable: true, - }; - let mut before_request = ModelRequest::new(vec![]).with_cache_segments(vec![segment.clone()]); - before_request.prompt_fingerprint = Some("before".into()); - let before = PromptCacheLayout::from_request(&before_request); - let mut after_request = ModelRequest::new(vec![]).with_cache_segments(vec![segment]); - after_request.prompt_fingerprint = Some("after".into()); - let after = PromptCacheLayout::from_request(&after_request); - assert!(!before.is_prefix_stable_against(&after)); - assert!(CacheLayoutEvent::new(&before, &after).changed_prefix); -} - -#[test] -fn cache_key_differs_for_different_requests() { - let r1 = ModelRequest::new(vec![]).with_model("gpt-4"); - let r2 = ModelRequest::new(vec![]).with_model("claude-3"); - assert_ne!(cache_key(&r1), cache_key(&r2)); -} - -#[test] -fn cache_key_is_deterministic_with_messages_and_tools() { - let build = || { - ModelRequest::new(vec![ - Message::system("you are terse"), - Message::user("hello"), - ]) - .with_model("gpt-4") - .with_tools(vec![ToolSchema::new( - "spin", - "spin a value", - json!({"type": "object"}), - )]) - }; - assert_eq!(cache_key(&build()), cache_key(&build())); -} - -#[test] -fn cache_key_reflects_message_content() { - let r1 = ModelRequest::new(vec![Message::user("hello")]); - let r2 = ModelRequest::new(vec![Message::user("goodbye")]); - assert_ne!( - cache_key(&r1), - cache_key(&r2), - "a changed message body must change the key" - ); -} - -#[test] -fn cache_key_reflects_message_count() { - let r1 = ModelRequest::new(vec![Message::user("hello")]); - let r2 = ModelRequest::new(vec![Message::user("hello"), Message::user("again")]); - assert_ne!( - cache_key(&r1), - cache_key(&r2), - "appending a message must change the key" - ); -} +use std::time::Duration; #[test] -fn cache_key_is_order_sensitive() { - let r1 = ModelRequest::new(vec![Message::user("a"), Message::user("b")]); - let r2 = ModelRequest::new(vec![Message::user("b"), Message::user("a")]); - assert_ne!( - cache_key(&r1), - cache_key(&r2), - "message order must change the key (length-prefixed frames)" - ); +fn cache_policy_defaults_to_no_caching() { + let policy = CachePolicy::default(); + assert!(!policy.response_cache_enabled); + assert!(!policy.protect_prompt_prefix); + assert_eq!(policy.ttl(), None); + assert_eq!(policy.namespace, None); } #[test] -fn cache_key_reflects_tool_schemas() { - let base = ModelRequest::new(vec![Message::user("hi")]); - let with_tool = base.clone().with_tools(vec![ToolSchema::new( - "spin", - "spin a value", - json!({"type": "object"}), - )]); - assert_ne!( - cache_key(&base), - cache_key(&with_tool), - "adding a tool schema must change the key" - ); +fn cache_policy_builders_set_fields() { + let policy = CachePolicy::enabled() + .with_ttl(Duration::from_secs(90)) + .with_namespace("tenant-7"); + assert!(policy.response_cache_enabled); + assert!(!policy.protect_prompt_prefix); + assert_eq!(policy.ttl(), Some(Duration::from_secs(90))); + assert_eq!(policy.namespace.as_deref(), Some("tenant-7")); } #[test] -fn cache_key_reflects_scalar_envelope_fields() { - let base = ModelRequest::new(vec![Message::user("hi")]); - let mut hot = base.clone(); - hot.temperature = Some(0.9); - assert_ne!( - cache_key(&base), - cache_key(&hot), - "a scalar field change must change the key via the envelope frame" +fn cache_policy_serde_omits_unset_options() { + let json = serde_json::to_value(CachePolicy::enabled()).unwrap(); + assert_eq!( + json, + json!({"response_cache_enabled": true, "protect_prompt_prefix": false}) ); + let back: CachePolicy = serde_json::from_value(json).unwrap(); + assert_eq!(back, CachePolicy::enabled()); } #[test] -fn cache_key_does_not_confuse_messages_with_tools() { - // A message array of length 1 and a tool array of length 1 are folded under - // distinct tags with count frames, so a request carrying one must not - // collide with an otherwise-empty request carrying the other. - let one_message = ModelRequest::new(vec![Message::user("x")]); - let one_tool = ModelRequest::new(vec![]).with_tools(vec![ToolSchema::new( - "x", - "x", - json!({"type": "object"}), - )]); - assert_ne!(cache_key(&one_message), cache_key(&one_tool)); -} - -#[test] -fn prompt_cache_layout_stable_prefix() { - let req = ModelRequest::new(vec![]).with_cache_segments(vec![ - PromptSegment { - id: "sys".into(), - role: SegmentRole::System, - cacheable: true, - }, - PromptSegment { - id: "tail".into(), - role: SegmentRole::Volatile, - cacheable: false, - }, - ]); - let layout = PromptCacheLayout::from_request(&req); - assert_eq!(layout.prefix_ids(), &["sys"]); - // Prompt-layout fingerprints are short local stability markers; response - // cache identity uses the stronger `cache_key` digest. - assert_eq!(layout.fingerprint().len(), 16); - - let same = PromptCacheLayout::from_request(&req); - assert!(layout.is_prefix_stable_against(&same)); -} - -#[test] -fn prompt_cache_layout_detects_changed_prefix() { - let req_a = ModelRequest::new(vec![]).with_cache_segments(vec![PromptSegment { - id: "sys".into(), - role: SegmentRole::System, - cacheable: true, - }]); - let req_b = ModelRequest::new(vec![]).with_cache_segments(vec![PromptSegment { - id: "sys-v2".into(), - role: SegmentRole::System, - cacheable: true, - }]); - - let before = PromptCacheLayout::from_request(&req_a); - let after = PromptCacheLayout::from_request(&req_b); - - assert!(!before.is_prefix_stable_against(&after)); - - let event = CacheLayoutEvent::new(&before, &after); - assert!(event.changed_prefix); - assert!(!event.volatile_only); - assert_eq!(event.segment_ids_before, vec!["sys"]); - assert_eq!(event.segment_ids_after, vec!["sys-v2"]); -} - -#[test] -fn cache_layout_event_volatile_only_when_no_cacheable_segments() { - let req_a = ModelRequest::new(vec![]).with_cache_segments(vec![PromptSegment { - id: "sys".into(), - role: SegmentRole::System, - cacheable: true, - }]); - let req_b = ModelRequest::new(vec![]).with_cache_segments(vec![PromptSegment { - id: "user".into(), - role: SegmentRole::Volatile, - cacheable: false, - }]); - - let before = PromptCacheLayout::from_request(&req_a); - let after = PromptCacheLayout::from_request(&req_b); - let event = CacheLayoutEvent::new(&before, &after); - - assert!(event.changed_prefix); - assert!( - event.volatile_only, - "all segments are volatile after the change" - ); +fn canonical_value_sorts_keys_at_every_depth() { + let value = json!({"b": {"z": 1, "a": [{"y": 2, "x": 3}]}, "a": null}); + let bytes = serde_json::to_string(&canonical_value(value)).unwrap(); + assert_eq!(bytes, r#"{"a":null,"b":{"a":[{"x":3,"y":2}],"z":1}}"#); } diff --git a/crates/tinyinference-llm/src/cache/types.rs b/crates/tinyinference-llm/src/cache/types.rs deleted file mode 100644 index 1587bf9b..00000000 --- a/crates/tinyinference-llm/src/cache/types.rs +++ /dev/null @@ -1,139 +0,0 @@ -//! Cache types for the harness cache module. -//! -//! These types let the recursive runtime answer a recurring request without -//! re-contacting the provider (response cache) and keep a provider's own -//! KV-cache prefix stable across the many requests a nested run produces -//! (layout protection). -//! -//! Two distinct caching concerns are modelled here: -//! -//! 1. **Local response cache** ([`ResponseCache`], [`InMemoryResponseCache`]): -//! a harness-side cache that lets the harness skip provider calls entirely -//! when the identical request has already been answered. -//! -//! 2. **Provider prompt / KV-cache layout protection** ([`PromptCacheLayout`], -//! [`CacheLayoutEvent`], [`CachePolicy`]): tooling for preserving the -//! stable byte-level prefix that the provider will cache in its own KV -//! store, without caching the actual response locally. -//! -//! All public types in this module are re-exported through [`super`]. - -use std::collections::{HashMap, VecDeque}; -use std::sync::{Arc, Mutex}; - -use async_trait::async_trait; -use serde::{Deserialize, Serialize}; - -use crate::Result; -use crate::model::ModelResponse; - -// ── ResponseCache ───────────────────────────────────────────────────────────── - -/// Local response cache that lets the harness skip provider calls entirely. -/// -/// Keys should be produced by [`super::cache_key`] for consistency. Callers -/// are responsible for deciding when caching is safe (e.g., not caching -/// side-effecting tool calls). -#[async_trait] -pub trait ResponseCache: Send + Sync { - /// Returns the cached [`ModelResponse`] for `key`, or `None` on a miss. - async fn get(&self, key: &str) -> Result>; - - /// Stores `value` under `key`. - async fn put(&self, key: &str, value: ModelResponse) -> Result<()>; -} - -/// Thread-safe in-memory response cache. -/// -/// Intended for unit tests and short-lived local runs. Contains no durable -/// storage: all entries are lost when the value is dropped. -/// -/// Entries are bounded by an LRU eviction policy (default -/// [`InMemoryResponseCache::DEFAULT_CAPACITY`]) so a long-lived cache attached -/// to a busy harness cannot grow without limit. Reads and writes move a key to -/// the most-recently-used end; once the map is full the least-recently-used key -/// is evicted on insert. -#[derive(Clone, Debug)] -pub struct InMemoryResponseCache { - pub(crate) inner: Arc>, -} - -/// LRU-ordered map backing [`InMemoryResponseCache`]. -#[derive(Debug)] -pub(crate) struct LruResponseMap { - /// Cached responses keyed by cache key. - pub(crate) data: HashMap, - /// Keys in least- to most-recently-used order. - pub(crate) order: VecDeque, - /// Maximum number of entries retained before LRU eviction. - pub(crate) capacity: usize, -} - -// ── PromptCacheLayout ───────────────────────────────────────────────────────── - -/// A snapshot of the ordered cacheable prompt-segment prefix that the provider -/// will see and may cache in its own KV store. -/// -/// The harness computes a `PromptCacheLayout` before and after each middleware -/// pass so it can detect and report accidental prefix invalidations. -/// -/// # Provider KV-cache stability rules -/// - Never insert timestamps, run ids, or dynamic retrieval output into the -/// stable prefix. -/// - Volatile content (latest user turn, tool results, scratchpads) should -/// always follow stable segments. -/// - Segment ordering must be preserved unless a middleware explicitly declares -/// a cache-layout migration. -#[derive(Clone, Debug, PartialEq, Eq)] -pub struct PromptCacheLayout { - /// Ordered ids of cacheable (stable) prefix segments. - pub(crate) prefix_ids: Vec, - /// Deterministic fingerprint of the ordered prefix ids. - pub(crate) fingerprint: String, - /// Caller-computed stable-prefix content fingerprint, when available. - pub(crate) content_fingerprint: Option, -} - -// ── CacheLayoutEvent ────────────────────────────────────────────────────────── - -/// Describes a change to the prompt cache layout that middleware can emit. -/// -/// Consumers (observability sinks, cost accounting, regression tests) can -/// inspect this struct to understand why a provider prompt-cache prefix was -/// preserved or invalidated. -#[derive(Clone, Debug)] -pub struct CacheLayoutEvent { - /// `true` if the cacheable prefix changed between `segment_ids_before` and - /// `segment_ids_after`. - pub changed_prefix: bool, - /// `true` if `segment_ids_after` contains only volatile (non-cacheable) - /// segments, meaning no stable prefix is present. - pub volatile_only: bool, - /// The ordered cacheable prefix ids before the middleware pass. - pub segment_ids_before: Vec, - /// The ordered cacheable prefix ids after the middleware pass. - pub segment_ids_after: Vec, -} - -// ── CachePolicy ─────────────────────────────────────────────────────────────── - -/// Policy knobs controlling both response caching and provider prompt-cache -/// layout protection. -/// -/// Both flags default to `false` (no caching / no protection) so the harness -/// is safe-by-default and opts must be explicit. -#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] -pub struct CachePolicy { - /// When `true`, the harness will look up (and write) local response cache - /// entries via [`ResponseCache`] before calling the provider. - pub response_cache_enabled: bool, - /// When `true`, middleware must preserve the order and content of cacheable - /// prefix segments. Violations are reported as [`CacheLayoutEvent`]s. - pub protect_prompt_prefix: bool, - /// Entry time-to-live in milliseconds; `None` means no expiry. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub ttl_ms: Option, - /// Optional cache-key namespace. - #[serde(default, skip_serializing_if = "Option::is_none")] - pub namespace: Option, -} From 950a86584a5461dff316fe7fb6a427011ac41355 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Wed, 7 Oct 2026 07:58:18 +0300 Subject: [PATCH 2/3] docs(changelog): note removal of stale response cache types Document the breaking removal of the duplicated harness response cache from tinyinference_llm::cache, since the maintained versions live in tinyagents-harness and nothing used the stale copies. Also note that canonical_value is now public and the crate no longer depends on sha2. Auto-committed-on: dragonfly Co-authored-by: Medulla --- CHANGELOG.md | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 020f8b2b..acb15036 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,16 @@ ## Unreleased +### Breaking changes + +- Removed the stale copy of the harness response cache from + `tinyinference_llm::cache`: `cache_key`, the `ResponseCache` trait, + `InMemoryResponseCache`, `PromptCacheLayout` and `CacheLayoutEvent`. Nothing + used them; the maintained versions (scoped keys, SQLite store, singleflight, + prompt-layout guard) live in `tinyagents-harness`'s `cache` module. + `CachePolicy` stays at `tinyinference_llm::cache::CachePolicy`, unchanged. + `canonical_value` is now public there. The crate no longer depends on `sha2`. + ### Added - `tinyinference-image`: the `ImageGenerator` trait, `OpenRouterImageGenerator` From 52278b0e37c8272d76725a2f2fc4e9a913e7ff76 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Wed, 7 Oct 2026 07:58:56 +0300 Subject: [PATCH 3/3] chore(deps): drop sha2 from the lockfile Remove the sha2 entry from the dependency list in Cargo.lock, reflecting that the crate is no longer a direct dependency. Auto-committed-on: dragonfly Co-authored-by: Medulla --- Cargo.lock | 1 - 1 file changed, 1 deletion(-) diff --git a/Cargo.lock b/Cargo.lock index 9cb659bd..3ca034ac 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1328,7 +1328,6 @@ dependencies = [ "reqwest", "serde", "serde_json", - "sha2", "tempfile", "thiserror", "tinyinference-core",