From a4b1696a79c8f4d7e714a62e6ea255730562826f Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Sat, 19 Sep 2026 01:23:15 +0000 Subject: [PATCH 01/10] feat(tokens): byte-based token estimator for tool-call events --- src/agentcat/modules/token_estimate.py | 98 ++++++++++++++++++++++++ tests/test_token_estimate.py | 102 +++++++++++++++++++++++++ 2 files changed, 200 insertions(+) create mode 100644 src/agentcat/modules/token_estimate.py create mode 100644 tests/test_token_estimate.py diff --git a/src/agentcat/modules/token_estimate.py b/src/agentcat/modules/token_estimate.py new file mode 100644 index 0000000..8ce5a09 --- /dev/null +++ b/src/agentcat/modules/token_estimate.py @@ -0,0 +1,98 @@ +"""Token estimates for tool-call events. + +One divisor, pinned byte-for-byte across the TypeScript, Python and Go SDKs: +``ceil(utf8_bytes / 3.5)``. Measured on production MCP responses, the inner +text runs 3.71 bytes per token on OpenAI tokenizers and about 3.15 on +Claude's; 3.5 splits the difference. The input side counts the raw arguments +the model emitted; the output side counts the content-block text the harness +feeds back. Nothing else counts. See the TypeScript repo's +docs/superpowers/specs/2026-09-19-sdk-token-estimates-design.md. +""" + +from __future__ import annotations + +import json +import math +from typing import Any + +TOKEN_ESTIMATE_BYTES_PER_TOKEN = 3.5 + +# The server's MAX_TOKEN_COUNT: the columns are int32. +_MAX_TOKEN_COUNT = 2_147_483_647 + + +def estimate_tokens(byte_count: int) -> int: + """``ceil(bytes / 3.5)``, 0 for nothing, clamped to the server's column.""" + if byte_count <= 0: + return 0 + return min( + math.ceil(byte_count / TOKEN_ESTIMATE_BYTES_PER_TOKEN), _MAX_TOKEN_COUNT + ) + + +def _utf8_len(text: str) -> int: + return len(text.encode("utf-8")) + + +def _compact_json_bytes(value: Any) -> int | None: + """UTF-8 length of the compact JSON: no spaces, no ASCII escaping. + + ``ensure_ascii=False`` matters: the default would spell ``café`` as + ``caf\\u00e9`` and count 20 bytes where TypeScript and Go count 16. + """ + try: + text = json.dumps( + value, separators=(",", ":"), ensure_ascii=False, default=str + ) + except Exception: + return None + return _utf8_len(text) + + +def estimate_input_tokens(arguments: Any) -> int | None: + """Tokens the model spent emitting the call: the raw arguments, injected + parameters included. None when there are no arguments to count. + + Never raises: this runs inside the customer's request, and a failure to + estimate must cost at most this field, never the tool's response. + """ + try: + if arguments is None: + return None + byte_count = _compact_json_bytes(arguments) + return None if byte_count is None else estimate_tokens(byte_count) + except Exception: + return None + + +def estimate_output_tokens(response: Any) -> int | None: + """Tokens the model reads back: the text of the content blocks. + + Falls back to the whole response when it carries no content list; None + when there is no response at all. Never raises, for the same reason as + ``estimate_input_tokens``. + """ + try: + if response is None: + return None + content = response.get("content") if isinstance(response, dict) else None + if not isinstance(content, list): + byte_count = _compact_json_bytes(response) + return None if byte_count is None else estimate_tokens(byte_count) + return estimate_tokens( + sum(_content_block_bytes(block) for block in content) + ) + except Exception: + return None + + +def _content_block_bytes(block: Any) -> int: + if not isinstance(block, dict): + return 0 + if block.get("type") == "text" and isinstance(block.get("text"), str): + return _utf8_len(block["text"]) + if block.get("type") == "resource": + resource = block.get("resource") + if isinstance(resource, dict) and isinstance(resource.get("text"), str): + return _utf8_len(resource["text"]) + return 0 diff --git a/tests/test_token_estimate.py b/tests/test_token_estimate.py new file mode 100644 index 0000000..ec25979 --- /dev/null +++ b/tests/test_token_estimate.py @@ -0,0 +1,102 @@ +"""Shared token-estimate vectors, pinned byte-for-byte against the TypeScript +and Go SDKs. See the TypeScript repo's +docs/superpowers/specs/2026-09-19-sdk-token-estimates-design.md.""" + +import pytest + +from agentcat.modules.token_estimate import ( + TOKEN_ESTIMATE_BYTES_PER_TOKEN, + estimate_input_tokens, + estimate_output_tokens, + estimate_tokens, +) + + +def text(t: str) -> dict: + return {"type": "text", "text": t} + + +def test_the_divisor_is_pinned(): + assert TOKEN_ESTIMATE_BYTES_PER_TOKEN == 3.5 + + +@pytest.mark.parametrize( + "byte_count, tokens", + [(0, 0), (1, 1), (7, 2), (13, 4), (4096, 1171), (7516192765, 2147483647)], +) +def test_estimate_tokens(byte_count, tokens): + assert estimate_tokens(byte_count) == tokens + + +@pytest.mark.parametrize( + "arguments, tokens", + [ + ({"q": "hello"}, 4), + ({}, 1), + ({"name": "café"}, 5), # 20 bytes -> 6 under ensure_ascii; must be 16 -> 5 + ({"html": "&"}, 6), + ({"t": "你好"}, 4), + ({"ids": [1, 2, 3], "opts": {"deep": True, "n": None}}, 13), + ], +) +def test_estimate_input_tokens(arguments, tokens): + assert estimate_input_tokens(arguments) == tokens + + +def test_absent_arguments_are_omitted(): + assert estimate_input_tokens(None) is None + + +def test_unserializable_arguments_fall_back_to_str(): + # default=str keeps an odd value countable rather than dropping the field. + assert estimate_input_tokens({"when": object()}) is not None + + +class _Hostile: + def __str__(self) -> str: + raise RuntimeError("boom") + + +class _HostileResponse(dict): + def get(self, *args, **kwargs): + raise RuntimeError("boom") + + +def test_never_raises_on_the_input_side(): + # default=str calls __str__, which raises: the field is omitted, not the call. + assert estimate_input_tokens({"when": _Hostile()}) is None + + +def test_never_raises_on_the_output_side(): + assert estimate_output_tokens(_HostileResponse(content=[])) is None + + +@pytest.mark.parametrize( + "response, tokens", + [ + ({"content": [text("hello world")]}, 4), + ({"content": [text("abcd"), text("e")]}, 2), + ({"content": [text("")]}, 0), + ({"content": [{"type": "image", "data": "QUJD", "mimeType": "image/png"}]}, 0), + ( + {"content": [{"type": "resource", "resource": {"uri": "file:///a", "text": "resource body"}}]}, + 4, + ), + ({"content": [{"type": "resource", "resource": {"uri": "file:///a", "blob": "QUJD"}}]}, 0), + ({"content": [text("hi")], "structuredContent": {"big": "y" * 1000}}, 1), + ({"content": [text("hi")], "structured_content": {"big": "y" * 1000}}, 1), + ({"content": [text("x" * 4096)]}, 1171), + ({"content": [text("hi")], "isError": True}, 1), + ({"content": [{"type": "text", "text": 42}, None, "str"]}, 0), + ], +) +def test_estimate_output_tokens(response, tokens): + assert estimate_output_tokens(response) == tokens + + +def test_no_content_list_falls_back_to_the_whole_response(): + assert estimate_output_tokens({"result": "ok"}) == 5 + + +def test_absent_response_is_omitted(): + assert estimate_output_tokens(None) is None From 4c20dbf6fc0f585f2b36424374196c57582488b8 Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Sat, 19 Sep 2026 01:34:53 +0000 Subject: [PATCH 02/10] feat(callpath): estimate input and output tokens before redaction --- 2026-09-19-cross-sdk-token-estimates.md | 76 +++++++++++++++++++ README.md | 7 ++ src/agentcat/modules/callpath.py | 6 ++ tests/test_token_estimates_integration.py | 92 +++++++++++++++++++++++ 4 files changed, 181 insertions(+) create mode 100644 2026-09-19-cross-sdk-token-estimates.md create mode 100644 tests/test_token_estimates_integration.py diff --git a/2026-09-19-cross-sdk-token-estimates.md b/2026-09-19-cross-sdk-token-estimates.md new file mode 100644 index 0000000..628856c --- /dev/null +++ b/2026-09-19-cross-sdk-token-estimates.md @@ -0,0 +1,76 @@ +# AgentCat SDK token estimates — Cross-SDK brief + +**Audience:** maintainers of the AgentCat TypeScript, Python and Go SDKs. +**Reference implementation:** `agentcat` (TypeScript) `src/modules/tokenEstimate.ts`. +**Server contract:** `input_tokens` / `output_tokens` on `PublishEventRequest` (agentcat-api 1.0.2), tagged `agentcat:token_source=sdk`. + +Every `mcp:tools/call` event carries two integers estimated by the SDK on the +raw payloads, before any redaction hook runs. The rules are pinned +byte-for-byte; do not re-derive them. + +## Estimator + +`estimate(bytes) = 0` when `bytes == 0`, else `ceil(bytes / 3.5)`, clamped to +`2147483647`. `bytes` is the UTF-8 byte length. The constant is +`TOKEN_ESTIMATE_BYTES_PER_TOKEN` (TypeScript, Python) / `BytesPerToken` (Go). + +## Input side: `input_tokens` + +The compact JSON of the raw, unstripped `arguments` object, injected +parameters included. Compact separators, no ASCII escaping +(`ensure_ascii=False`), no HTML escaping (`SetEscapeHTML(false)`), no trailing +newline. Not the tool name, `_meta`, `extra`, or the JSON-RPC envelope. +Absent or null arguments → field omitted. `{}` → 1. + +## Output side: `output_tokens` + +The summed UTF-8 bytes of `text` blocks and string `resource.text` values in +`content`, divided once. Image, audio, blob and unknown blocks count 0. Not +`structuredContent`, `isError`, `_meta`, the envelope, or the mint-back text. +No `content` list → the compact JSON of the whole recorded response. Absent +response → field omitted. Content with no text-bearing block → 0. + +## Ordering + +Both values are set in the tool-call wrapper on the raw objects, before the +queue runs `redact_event` → `redact_sensitive_information` → sanitize → +truncate. Nothing recomputes them afterwards. + +## Vectors + +| Arguments | Bytes | Tokens | +|---|---|---| +| `{"q":"hello"}` | 13 | 4 | +| `{}` | 2 | 1 | +| `{"name":"café"}` | 16 | 5 | +| `{"html":"&"}` | 19 | 6 | +| `{"t":"你好"}` | 14 | 4 | +| `{"ids":[1,2,3],"opts":{"deep":true,"n":null}}` | 45 | 13 | + +| Content | Bytes | Tokens | +|---|---|---| +| text `"hello world"` | 11 | 4 | +| text `"abcd"` + text `"e"` | 5 | 2 | +| text `""` | 0 | 0 | +| image block | 0 | 0 | +| resource with text `"resource body"` | 13 | 4 | +| resource with blob only | 0 | 0 | +| text `"hi"` + 1000-byte `structuredContent` | 2 | 1 | +| `{"result":"ok"}` (no content) | 15 | 5 | +| text of 4096 `x` | 4096 | 1171 | + +| Bytes | Tokens | +|---|---| +| 0 | 0 | +| 1 | 1 | +| 7 | 2 | +| 7516192765 | 2147483647 | + +## Measurement basis + +1,240 production responses (2026-09-12 to 2026-09-19), tokenized with +tiktoken `cl100k_base` and `o200k_base`: inner text 3.71 bytes per token +weighted (p10 3.13, p50 3.75, p90 4.64), the two encodings within 0.2 percent; +JSON 3.72, prose 4.52; the serialized envelope inflates bytes and tokens by +1.19×. Claude-family tokenizers use roughly 15 to 20 percent more tokens +(Anthropic guidance, not yet measured); 3.5 splits the difference. diff --git a/README.md b/README.md index cde0c07..3b08ad6 100644 --- a/README.md +++ b/README.md @@ -162,6 +162,13 @@ def redact_event(event): agentcat.track(server, "proj_0000000", AgentCatOptions(redact_event=redact_event)) ``` +Tool-call events also carry `input_tokens` and `output_tokens`: estimates +(`ceil(utf8_bytes / 3.5)`) of the raw tool arguments and of the text in the +result's content blocks. They are computed before your redaction hooks run, +so they describe the original payload size even when the stored strings are +redacted. Drop the two fields in `redact_event` if that size must not leave +your server. + ### Vendor Support AgentCat seamlessly integrates with your existing observability stack, providing automatic logging and tracing without the tedious setup typically required. Export telemetry data to multiple platforms simultaneously: diff --git a/src/agentcat/modules/callpath.py b/src/agentcat/modules/callpath.py index 6d24eb8..71a2300 100644 --- a/src/agentcat/modules/callpath.py +++ b/src/agentcat/modules/callpath.py @@ -56,6 +56,8 @@ UserIdentity, ) +from .token_estimate import estimate_input_tokens, estimate_output_tokens + # A failed call whose adapter could make nothing of the failure. Same shape as # every other error payload so consumers never have to branch on presence. _UNKNOWN_ERROR: ErrorData = { @@ -305,6 +307,10 @@ async def publish_tool_call_event( is_error=is_error, error=(error or _UNKNOWN_ERROR) if is_error else None, duration=duration_ms, + # Estimated on the raw payloads here, before the queue's redaction + # hooks run; the queue never recomputes them. + input_tokens=estimate_input_tokens(raw_arguments), + output_tokens=estimate_output_tokens(response), client_name=rc.client.name, client_version=rc.client.version, identify_actor_given_id=rc.actor.user_id if rc.actor else None, diff --git a/tests/test_token_estimates_integration.py b/tests/test_token_estimates_integration.py new file mode 100644 index 0000000..c6050a4 --- /dev/null +++ b/tests/test_token_estimates_integration.py @@ -0,0 +1,92 @@ +"""Token estimates ride on every tools/call event, computed on the raw +payloads before the queue's redaction and truncation stages run. + +Vectors: {"text":"hi there"} is 19 bytes -> 6 tokens; the flavors' `echo` +tool answers "echo:hi there", 13 bytes -> 4. {"text":"secret-value-123456789"} +is 33 bytes -> 10, and the echo of it, "echo:secret-value-123456789", is 27 +bytes -> 8. +""" + +import pytest + +from agentcat import AgentCatOptions, track +from agentcat.modules import event_queue + +from .test_utils.flavors import flavors + + +@pytest.fixture(autouse=True) +def capture(monkeypatch): + """Collect every event the queue is handed, without touching the network.""" + events: list = [] + monkeypatch.setattr(event_queue.event_queue, "add", events.append) + return events + + +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_tool_call_events_carry_token_estimates(flavor, capture): + built = flavor.build("token-estimates") + track(built.server, "proj_test", AgentCatOptions()) + + async with flavor.client(built.server) as client: + await flavor.list_tools(client) + await flavor.call(client, "echo", {"text": "hi there"}) + + event = capture[0] + assert event.input_tokens == 6 + # Only the content text counts: the structured mirror of the same answer + # and the era-specific error/structured keys add nothing. + assert event.output_tokens == 4 + + +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_counts_survive_redaction_and_truncation(flavor, capture, monkeypatch): + sent: list = [] + monkeypatch.setattr(event_queue.event_queue, "_send_event", sent.append) + + built = flavor.build("token-estimates-redacted") + track( + built.server, + "proj_test", + AgentCatOptions(redact_sensitive_information=lambda _text: "[REDACTED]"), + ) + + async with flavor.client(built.server) as client: + await flavor.list_tools(client) + await flavor.call(client, "echo", {"text": "secret-value-123456789"}) + + queued = capture[0] + assert queued.redaction_fn is not None + # Drive the real pipeline (redact -> sanitize -> truncate -> send) on the + # captured event, exactly as the worker thread would. + event_queue.event_queue._process_event(queued) + + assert len(sent) == 1 + published = sent[0] + assert published.parameters["arguments"]["text"] == "[REDACTED]" + assert published.response["content"][0]["text"] == "[REDACTED]" + assert published.input_tokens == 10 + assert published.output_tokens == 8 # "echo:secret-value-123456789" = 27 bytes + + +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_a_broken_estimator_never_breaks_the_tool_call(flavor, capture, monkeypatch): + """The estimator runs inside the customer's request. Force it to raise + and the client must still receive the tool's own answer; the call's + analytics may be lost, the response never is.""" + + def explode(_value): + raise RuntimeError("estimator bug") + + monkeypatch.setattr("agentcat.modules.callpath.estimate_input_tokens", explode) + monkeypatch.setattr("agentcat.modules.callpath.estimate_output_tokens", explode) + + built = flavor.build("token-estimates-broken") + track(built.server, "proj_test", AgentCatOptions()) + + async with flavor.client(built.server) as client: + await flavor.list_tools(client) + called = await flavor.call(client, "echo", {"text": "hi there"}) + + assert not called.is_error + assert "echo:hi there" in called.text From bda28641217b4a99c7723e3df6b621708a0b2ac4 Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Sat, 19 Sep 2026 01:38:46 +0000 Subject: [PATCH 03/10] test(tokens): redact only the secret string in the ordering test --- tests/test_token_estimates_integration.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/tests/test_token_estimates_integration.py b/tests/test_token_estimates_integration.py index c6050a4..b8eb017 100644 --- a/tests/test_token_estimates_integration.py +++ b/tests/test_token_estimates_integration.py @@ -48,7 +48,13 @@ async def test_counts_survive_redaction_and_truncation(flavor, capture, monkeypa track( built.server, "proj_test", - AgentCatOptions(redact_sensitive_information=lambda _text: "[REDACTED]"), + # Only the secret string is rewritten: a hook that replaced every + # string would also clobber the content block's `type` discriminator. + AgentCatOptions( + redact_sensitive_information=lambda text: ( + "[REDACTED]" if "secret" in text else text + ) + ), ) async with flavor.client(built.server) as client: From e149dd2a24afd4fa55b4debd90eff796154303cc Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Sat, 19 Sep 2026 02:04:20 +0000 Subject: [PATCH 04/10] chore(release): 2.1.1 --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 024038e..567acff 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "agentcat" -version = "2.1.0" +version = "2.1.1" description = "Analytics tool for MCP (Model Context Protocol) servers, Claude Connectors, and ChatGPT Plugins - tracks tool usage patterns and provides insights" authors = [ { name = "AgentCat, Inc.", email = "support@agentcat.com" }, From 53bb7d492db9a93474082ae9565187bfeec8f0f4 Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Sat, 19 Sep 2026 02:32:07 +0000 Subject: [PATCH 05/10] fix(tokens): match parity findings from the final SDK token-estimates review Drop json.dumps's default=str so an unserializable argument is omitted rather than counted (TypeScript and Go already omit it), and encode with surrogatepass so a lone surrogate off the wire counts its 3 bytes instead of dropping the whole field. Add coverage for both, add a truncation leg to the tools/call ordering test, and note the two Go-only edge cases (empty-object counting, isError responses) plus serialization corner cases in the cross-SDK brief. --- 2026-09-19-cross-sdk-token-estimates.md | 16 ++++++++-- src/agentcat/modules/token_estimate.py | 16 ++++++---- tests/test_token_estimate.py | 24 ++++++++++++--- tests/test_token_estimates_integration.py | 36 +++++++++++++++++++++++ 4 files changed, 81 insertions(+), 11 deletions(-) diff --git a/2026-09-19-cross-sdk-token-estimates.md b/2026-09-19-cross-sdk-token-estimates.md index 628856c..e65cded 100644 --- a/2026-09-19-cross-sdk-token-estimates.md +++ b/2026-09-19-cross-sdk-token-estimates.md @@ -20,7 +20,10 @@ The compact JSON of the raw, unstripped `arguments` object, injected parameters included. Compact separators, no ASCII escaping (`ensure_ascii=False`), no HTML escaping (`SetEscapeHTML(false)`), no trailing newline. Not the tool name, `_meta`, `extra`, or the JSON-RPC envelope. -Absent or null arguments → field omitted. `{}` → 1. +Absent or null arguments → field omitted. `{}` → 1. Adapters that cannot +tell absent arguments from an empty object (Python's three adapters and the +Go official-SDK adapter, which hand the funnel a map either way) count `{}` +→ 1; TypeScript and Go mcp-go omit. ## Output side: `output_tokens` @@ -28,7 +31,16 @@ The summed UTF-8 bytes of `text` blocks and string `resource.text` values in `content`, divided once. Image, audio, blob and unknown blocks count 0. Not `structuredContent`, `isError`, `_meta`, the envelope, or the mint-back text. No `content` list → the compact JSON of the whole recorded response. Absent -response → field omitted. Content with no text-bearing block → 0. +response → field omitted. Content with no text-bearing block → 0. The Go +adapters record no `Response` on `isError` results, so Go omits +`output_tokens` there while TypeScript and Python count the error text. + +## Serialization corner cases + +Non-canonical numbers (`1.0`; integers at or above 1e21, which +`JSON.stringify` writes as `1e+21`) and U+2028/U+2029 (Go always escapes +them to six bytes) can differ by a byte or two between SDKs, all within ±1 +token. Lone surrogates count 3 bytes in every SDK. ## Ordering diff --git a/src/agentcat/modules/token_estimate.py b/src/agentcat/modules/token_estimate.py index 8ce5a09..28312c3 100644 --- a/src/agentcat/modules/token_estimate.py +++ b/src/agentcat/modules/token_estimate.py @@ -31,19 +31,25 @@ def estimate_tokens(byte_count: int) -> int: def _utf8_len(text: str) -> int: - return len(text.encode("utf-8")) + # surrogatepass: a lone surrogate off the wire (json.loads on a string + # containing an unpaired \udXXX escape) must still be counted, not + # dropped. TypeScript and Go's decoders substitute U+FFFD (3 bytes) for + # the same input; surrogatepass encodes a lone surrogate to the same 3 + # bytes, so the byte count matches across SDKs. + return len(text.encode("utf-8", errors="surrogatepass")) def _compact_json_bytes(value: Any) -> int | None: """UTF-8 length of the compact JSON: no spaces, no ASCII escaping. ``ensure_ascii=False`` matters: the default would spell ``café`` as - ``caf\\u00e9`` and count 20 bytes where TypeScript and Go count 16. + ``caf\\u00e9`` and count 20 bytes where TypeScript and Go count 16. No + ``default=str``: an unserializable value must be omitted, matching + TypeScript and Go, which also drop the field rather than counting a + stand-in string. """ try: - text = json.dumps( - value, separators=(",", ":"), ensure_ascii=False, default=str - ) + text = json.dumps(value, separators=(",", ":"), ensure_ascii=False) except Exception: return None return _utf8_len(text) diff --git a/tests/test_token_estimate.py b/tests/test_token_estimate.py index ec25979..b510883 100644 --- a/tests/test_token_estimate.py +++ b/tests/test_token_estimate.py @@ -47,9 +47,18 @@ def test_absent_arguments_are_omitted(): assert estimate_input_tokens(None) is None -def test_unserializable_arguments_fall_back_to_str(): - # default=str keeps an odd value countable rather than dropping the field. - assert estimate_input_tokens({"when": object()}) is not None +def test_lone_surrogate_counts_three_bytes_on_the_input_side(): + # A lone surrogate off the wire (json.loads on an unpaired \ud83d escape) + # must count, not be omitted: surrogatepass encodes it to 3 bytes, the + # same width Go counts after its decoder substitutes U+FFFD. + # {"a":""} = 6 + 3 + 2 = 11 bytes -> ceil(11 / 3.5) = 4. + assert estimate_input_tokens({"a": "\ud83d"}) == 4 + + +def test_unserializable_arguments_are_omitted(): + # No default=str: TypeScript and Go both omit an unserializable value + # rather than counting a stand-in string, so Python must match. + assert estimate_input_tokens({"when": object()}) is None class _Hostile: @@ -63,7 +72,9 @@ def get(self, *args, **kwargs): def test_never_raises_on_the_input_side(): - # default=str calls __str__, which raises: the field is omitted, not the call. + # json.dumps raises TypeError on an unserializable value before __str__ + # is ever consulted (no default=str); the field is omitted either way, + # so a hostile __str__ never gets the chance to raise into the call. assert estimate_input_tokens({"when": _Hostile()}) is None @@ -98,5 +109,10 @@ def test_no_content_list_falls_back_to_the_whole_response(): assert estimate_output_tokens({"result": "ok"}) == 5 +def test_lone_surrogate_counts_three_bytes_on_the_output_side(): + # A lone surrogate is 3 bytes -> ceil(3 / 3.5) = 1. + assert estimate_output_tokens({"content": [text("\ud83d")]}) == 1 + + def test_absent_response_is_omitted(): assert estimate_output_tokens(None) is None diff --git a/tests/test_token_estimates_integration.py b/tests/test_token_estimates_integration.py index b8eb017..eab9a5f 100644 --- a/tests/test_token_estimates_integration.py +++ b/tests/test_token_estimates_integration.py @@ -75,6 +75,42 @@ async def test_counts_survive_redaction_and_truncation(flavor, capture, monkeypa assert published.output_tokens == 8 # "echo:secret-value-123456789" = 27 bytes +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_counts_survive_truncation(flavor, capture, monkeypatch): + """{"text":"xxx...x"} (60000 x's) is 60011 bytes -> ceil(60011/3.5) = 17146 + exactly. The echo of it, "echo:" + 60000 x's, is 60005 bytes -> + ceil(60005/3.5) = 17145. The brief's original 40000-x vector (11432 / + 11430) does not reliably push every flavor's serialized event past + truncation.MAX_EVENT_BYTES (100KB) — some flavors mirror the answer into + `structuredContent` too and some do not, so only the larger payload + guarantees size-targeted truncation fires on every flavor. The published + response text ends up well below the original 60005 characters, and the + counts still describe the original, untruncated bytes. + """ + sent: list = [] + monkeypatch.setattr(event_queue.event_queue, "_send_event", sent.append) + + built = flavor.build("token-estimates-truncated") + track(built.server, "proj_test", AgentCatOptions()) + + long_text = "x" * 60000 + async with flavor.client(built.server) as client: + await flavor.list_tools(client) + await flavor.call(client, "echo", {"text": long_text}) + + queued = capture[0] + # Drive the real pipeline (sanitize -> truncate -> send) on the captured + # event, exactly as the worker thread would. + event_queue.event_queue._process_event(queued) + + assert len(sent) == 1 + published = sent[0] + published_text = published.response["content"][0]["text"] + assert len(published_text) < 60005 + assert published.input_tokens == 17146 + assert published.output_tokens == 17145 + + @pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) async def test_a_broken_estimator_never_breaks_the_tool_call(flavor, capture, monkeypatch): """The estimator runs inside the customer's request. Force it to raise From 0aa529415463973999ece1112443eaa515756ced Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Sat, 19 Sep 2026 08:05:13 +0000 Subject: [PATCH 06/10] feat(tokens): count structured content when a result has no content blocks --- 2026-09-19-cross-sdk-token-estimates.md | 10 +++ README.md | 3 +- src/agentcat/modules/token_estimate.py | 8 ++ tests/test_token_estimate.py | 11 +++ tests/test_token_estimates_integration.py | 96 +++++++++++++++++++++++ 5 files changed, 127 insertions(+), 1 deletion(-) diff --git a/2026-09-19-cross-sdk-token-estimates.md b/2026-09-19-cross-sdk-token-estimates.md index e65cded..fd0cbe5 100644 --- a/2026-09-19-cross-sdk-token-estimates.md +++ b/2026-09-19-cross-sdk-token-estimates.md @@ -35,6 +35,13 @@ response → field omitted. Content with no text-bearing block → 0. The Go adapters record no `Response` on `isError` results, so Go omits `output_tokens` there while TypeScript and Python count the error text. +Structured-only results: when `content` is present and EMPTY and the result +carries `structuredContent` (Python: also the `structured_content` spelling), +`output_tokens` is the compact JSON of that structured value instead of 0. Any +non-empty `content`, even image-only, keeps the text-only count; a null or +absent structured value on empty content still counts 0. Vector: +`{"content":[],"structuredContent":{"result":"ok"}}` → 15 bytes → 5. + ## Serialization corner cases Non-canonical numbers (`1.0`; integers at or above 1e21, which @@ -69,6 +76,9 @@ truncate. Nothing recomputes them afterwards. | resource with blob only | 0 | 0 | | text `"hi"` + 1000-byte `structuredContent` | 2 | 1 | | `{"result":"ok"}` (no content) | 15 | 5 | +| empty content + `structuredContent` `{"result":"ok"}` | 15 | 5 (structured-only) | +| empty content, no `structuredContent` | 0 | 0 | +| image-only content + `structuredContent` `{"result":"ok"}` | 0 | 0 | | text of 4096 `x` | 4096 | 1171 | | Bytes | Tokens | diff --git a/README.md b/README.md index 3b08ad6..79b3ce2 100644 --- a/README.md +++ b/README.md @@ -164,7 +164,8 @@ agentcat.track(server, "proj_0000000", AgentCatOptions(redact_event=redact_event Tool-call events also carry `input_tokens` and `output_tokens`: estimates (`ceil(utf8_bytes / 3.5)`) of the raw tool arguments and of the text in the -result's content blocks. They are computed before your redaction hooks run, +result's content blocks (or of the structured content, when a result has no +content blocks). They are computed before your redaction hooks run, so they describe the original payload size even when the stored strings are redacted. Drop the two fields in `redact_event` if that size must not leave your server. diff --git a/src/agentcat/modules/token_estimate.py b/src/agentcat/modules/token_estimate.py index 28312c3..72b81d2 100644 --- a/src/agentcat/modules/token_estimate.py +++ b/src/agentcat/modules/token_estimate.py @@ -85,6 +85,14 @@ def estimate_output_tokens(response: Any) -> int | None: if not isinstance(content, list): byte_count = _compact_json_bytes(response) return None if byte_count is None else estimate_tokens(byte_count) + if len(content) == 0: + structured = response.get("structuredContent") + if structured is None: + structured = response.get("structured_content") + if structured is not None: + byte_count = _compact_json_bytes(structured) + return None if byte_count is None else estimate_tokens(byte_count) + return 0 return estimate_tokens( sum(_content_block_bytes(block) for block in content) ) diff --git a/tests/test_token_estimate.py b/tests/test_token_estimate.py index b510883..821cf0b 100644 --- a/tests/test_token_estimate.py +++ b/tests/test_token_estimate.py @@ -99,6 +99,17 @@ def test_never_raises_on_the_output_side(): ({"content": [text("x" * 4096)]}, 1171), ({"content": [text("hi")], "isError": True}, 1), ({"content": [{"type": "text", "text": 42}, None, "str"]}, 0), + ({"content": [], "structuredContent": {"result": "ok"}}, 5), + ({"content": []}, 0), + ({"content": [], "structuredContent": None}, 0), + ( + { + "content": [{"type": "image", "data": "QUJD", "mimeType": "image/png"}], + "structuredContent": {"result": "ok"}, + }, + 0, + ), + ({"content": [], "structured_content": {"result": "ok"}}, 5), ], ) def test_estimate_output_tokens(response, tokens): diff --git a/tests/test_token_estimates_integration.py b/tests/test_token_estimates_integration.py index eab9a5f..37f41b0 100644 --- a/tests/test_token_estimates_integration.py +++ b/tests/test_token_estimates_integration.py @@ -39,6 +39,102 @@ async def test_tool_call_events_carry_token_estimates(flavor, capture): assert event.output_tokens == 4 +def _build_structured_only_server(flavor_id: str): + """A fresh, untracked server whose lone tool answers with an empty + ``content`` list and a ``structuredContent``/``structured_content`` of + ``{"result": "ok"}`` — no auto-mirrored text block. + + Each era's facade normally derives a text content block from a typed + return value, so this bypasses that conversion the way the era itself + allows: ``MCPServer`` and the community ``fastmcp`` both pass a + already-built result object straight through their `convert_result` + (`mcp.server.mcpserver.utilities.func_metadata.FuncMetadata.convert_result`, + `fastmcp.tools.base.Tool.convert_result`) instead of re-deriving content + from it, and the lowlevel `Server`'s `on_call_tool` callback is returned + to the wire completely unmodified. + """ + if flavor_id == "mcpserver-v2": + from mcp.server.mcpserver import MCPServer + from mcp.types import CallToolResult + + server = MCPServer("structured-only") + + @server.tool() + async def structured_only() -> CallToolResult: + return CallToolResult(content=[], structured_content={"result": "ok"}) + + return server + + if flavor_id == "lowlevel-v2": + from mcp import types + from mcp.server import Server + + async def on_list_tools(ctx, params): + return types.ListToolsResult( + tools=[ + types.Tool( + name="structured_only", + description="", + input_schema={"type": "object", "properties": {}}, + ) + ] + ) + + async def on_call_tool(ctx, params): + return types.CallToolResult(content=[], structured_content={"result": "ok"}) + + return Server( + "structured-only", on_list_tools=on_list_tools, on_call_tool=on_call_tool + ) + + if flavor_id.startswith("community-"): + from fastmcp import FastMCP + from fastmcp.tools.base import ToolResult + + server = FastMCP("structured-only") + + @server.tool + async def structured_only() -> ToolResult: + return ToolResult(content=[], structured_content={"result": "ok"}) + + return server + + return None + + +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_structured_only_result_counts_the_structured_content(flavor, capture): + """When a result's `content` list is empty and it carries a structured + value, `output_tokens` counts that value's compact JSON instead of 0. + + Not every flavor's facade can be made to answer with an empty `content` + list without going around its typed-return conversion; flavors that + cannot are skipped here rather than faked, per the design brief. + """ + server = _build_structured_only_server(flavor.id) + if server is None: + pytest.skip( + f"{flavor.id}: no known way to make this flavor answer with an " + "empty content list and a structured value" + ) + track(server, "proj_test", AgentCatOptions()) + + async with flavor.client(server) as client: + await flavor.call(client, "structured_only", {}) + + event = capture[0] + # Confirms the fixture itself, not just the estimate: the event records + # the customer's undecorated result, an empty content list with the + # structured value intact — the wire response the client actually sees + # also carries the SDK's session mint-back text, which must not count. + assert event.response["content"] == [] + structured = event.response.get("structured_content") or event.response.get( + "structuredContent" + ) + assert structured == {"result": "ok"} + assert event.output_tokens == 5 # {"result":"ok"} = 15 bytes + + @pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) async def test_counts_survive_redaction_and_truncation(flavor, capture, monkeypatch): sent: list = [] From 7828aa6181fb727c1d833457428954b66b504779 Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Tue, 22 Sep 2026 21:33:15 +0000 Subject: [PATCH 07/10] chore(deps): agentcat-api 1.0.2 for the token count fields --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 567acff..98c1b6d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -19,7 +19,7 @@ classifiers = [ ] dependencies = [ "mcp>=1.2.0,<3", - "agentcat-api==1.0.0", + "agentcat-api==1.0.2", "pydantic>=2.0.0,<3", "requests>=2.31.0", ] From 537df952cfcffc76fd6025e7da3381f5f07562c1 Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Tue, 22 Sep 2026 21:44:17 +0000 Subject: [PATCH 08/10] test(tokens): import ToolResult from fastmcp.tools for FastMCP 3.0 and 3.1 --- tests/test_token_estimates_integration.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/tests/test_token_estimates_integration.py b/tests/test_token_estimates_integration.py index 37f41b0..a2cca9c 100644 --- a/tests/test_token_estimates_integration.py +++ b/tests/test_token_estimates_integration.py @@ -49,7 +49,7 @@ def _build_structured_only_server(flavor_id: str): allows: ``MCPServer`` and the community ``fastmcp`` both pass a already-built result object straight through their `convert_result` (`mcp.server.mcpserver.utilities.func_metadata.FuncMetadata.convert_result`, - `fastmcp.tools.base.Tool.convert_result`) instead of re-deriving content + the community `Tool.convert_result`) instead of re-deriving content from it, and the lowlevel `Server`'s `on_call_tool` callback is returned to the wire completely unmodified. """ @@ -89,7 +89,11 @@ async def on_call_tool(ctx, params): if flavor_id.startswith("community-"): from fastmcp import FastMCP - from fastmcp.tools.base import ToolResult + + # `fastmcp.tools` re-exports ToolResult in every supported release; + # the module behind it moved (`fastmcp.tools.tool` through 3.1.x, + # `fastmcp.tools.base` from 3.2), so import from the package. + from fastmcp.tools import ToolResult server = FastMCP("structured-only") From 6a41a84b36df5027ef33e69da01194c01949d218 Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Tue, 29 Sep 2026 17:09:54 +0000 Subject: [PATCH 09/10] docs: drop the README note about token estimates --- README.md | 8 -------- 1 file changed, 8 deletions(-) diff --git a/README.md b/README.md index 79b3ce2..cde0c07 100644 --- a/README.md +++ b/README.md @@ -162,14 +162,6 @@ def redact_event(event): agentcat.track(server, "proj_0000000", AgentCatOptions(redact_event=redact_event)) ``` -Tool-call events also carry `input_tokens` and `output_tokens`: estimates -(`ceil(utf8_bytes / 3.5)`) of the raw tool arguments and of the text in the -result's content blocks (or of the structured content, when a result has no -content blocks). They are computed before your redaction hooks run, -so they describe the original payload size even when the stored strings are -redacted. Drop the two fields in `redact_event` if that size must not leave -your server. - ### Vendor Support AgentCat seamlessly integrates with your existing observability stack, providing automatic logging and tracing without the tedious setup typically required. Export telemetry data to multiple platforms simultaneously: From 1c9fe41e829747358cad14cd89d63840293bd60a Mon Sep 17 00:00:00 2001 From: Naseem Alnaji Date: Tue, 29 Sep 2026 17:32:04 +0000 Subject: [PATCH 10/10] docs: drop the cross-SDK token estimates brief --- 2026-09-19-cross-sdk-token-estimates.md | 98 ------------------------- 1 file changed, 98 deletions(-) delete mode 100644 2026-09-19-cross-sdk-token-estimates.md diff --git a/2026-09-19-cross-sdk-token-estimates.md b/2026-09-19-cross-sdk-token-estimates.md deleted file mode 100644 index fd0cbe5..0000000 --- a/2026-09-19-cross-sdk-token-estimates.md +++ /dev/null @@ -1,98 +0,0 @@ -# AgentCat SDK token estimates — Cross-SDK brief - -**Audience:** maintainers of the AgentCat TypeScript, Python and Go SDKs. -**Reference implementation:** `agentcat` (TypeScript) `src/modules/tokenEstimate.ts`. -**Server contract:** `input_tokens` / `output_tokens` on `PublishEventRequest` (agentcat-api 1.0.2), tagged `agentcat:token_source=sdk`. - -Every `mcp:tools/call` event carries two integers estimated by the SDK on the -raw payloads, before any redaction hook runs. The rules are pinned -byte-for-byte; do not re-derive them. - -## Estimator - -`estimate(bytes) = 0` when `bytes == 0`, else `ceil(bytes / 3.5)`, clamped to -`2147483647`. `bytes` is the UTF-8 byte length. The constant is -`TOKEN_ESTIMATE_BYTES_PER_TOKEN` (TypeScript, Python) / `BytesPerToken` (Go). - -## Input side: `input_tokens` - -The compact JSON of the raw, unstripped `arguments` object, injected -parameters included. Compact separators, no ASCII escaping -(`ensure_ascii=False`), no HTML escaping (`SetEscapeHTML(false)`), no trailing -newline. Not the tool name, `_meta`, `extra`, or the JSON-RPC envelope. -Absent or null arguments → field omitted. `{}` → 1. Adapters that cannot -tell absent arguments from an empty object (Python's three adapters and the -Go official-SDK adapter, which hand the funnel a map either way) count `{}` -→ 1; TypeScript and Go mcp-go omit. - -## Output side: `output_tokens` - -The summed UTF-8 bytes of `text` blocks and string `resource.text` values in -`content`, divided once. Image, audio, blob and unknown blocks count 0. Not -`structuredContent`, `isError`, `_meta`, the envelope, or the mint-back text. -No `content` list → the compact JSON of the whole recorded response. Absent -response → field omitted. Content with no text-bearing block → 0. The Go -adapters record no `Response` on `isError` results, so Go omits -`output_tokens` there while TypeScript and Python count the error text. - -Structured-only results: when `content` is present and EMPTY and the result -carries `structuredContent` (Python: also the `structured_content` spelling), -`output_tokens` is the compact JSON of that structured value instead of 0. Any -non-empty `content`, even image-only, keeps the text-only count; a null or -absent structured value on empty content still counts 0. Vector: -`{"content":[],"structuredContent":{"result":"ok"}}` → 15 bytes → 5. - -## Serialization corner cases - -Non-canonical numbers (`1.0`; integers at or above 1e21, which -`JSON.stringify` writes as `1e+21`) and U+2028/U+2029 (Go always escapes -them to six bytes) can differ by a byte or two between SDKs, all within ±1 -token. Lone surrogates count 3 bytes in every SDK. - -## Ordering - -Both values are set in the tool-call wrapper on the raw objects, before the -queue runs `redact_event` → `redact_sensitive_information` → sanitize → -truncate. Nothing recomputes them afterwards. - -## Vectors - -| Arguments | Bytes | Tokens | -|---|---|---| -| `{"q":"hello"}` | 13 | 4 | -| `{}` | 2 | 1 | -| `{"name":"café"}` | 16 | 5 | -| `{"html":"&"}` | 19 | 6 | -| `{"t":"你好"}` | 14 | 4 | -| `{"ids":[1,2,3],"opts":{"deep":true,"n":null}}` | 45 | 13 | - -| Content | Bytes | Tokens | -|---|---|---| -| text `"hello world"` | 11 | 4 | -| text `"abcd"` + text `"e"` | 5 | 2 | -| text `""` | 0 | 0 | -| image block | 0 | 0 | -| resource with text `"resource body"` | 13 | 4 | -| resource with blob only | 0 | 0 | -| text `"hi"` + 1000-byte `structuredContent` | 2 | 1 | -| `{"result":"ok"}` (no content) | 15 | 5 | -| empty content + `structuredContent` `{"result":"ok"}` | 15 | 5 (structured-only) | -| empty content, no `structuredContent` | 0 | 0 | -| image-only content + `structuredContent` `{"result":"ok"}` | 0 | 0 | -| text of 4096 `x` | 4096 | 1171 | - -| Bytes | Tokens | -|---|---| -| 0 | 0 | -| 1 | 1 | -| 7 | 2 | -| 7516192765 | 2147483647 | - -## Measurement basis - -1,240 production responses (2026-09-12 to 2026-09-19), tokenized with -tiktoken `cl100k_base` and `o200k_base`: inner text 3.71 bytes per token -weighted (p10 3.13, p50 3.75, p90 4.64), the two encodings within 0.2 percent; -JSON 3.72, prose 4.52; the serialized envelope inflates bytes and tokens by -1.19×. Claude-family tokenizers use roughly 15 to 20 percent more tokens -(Anthropic guidance, not yet measured); 3.5 splits the difference.