diff --git a/pyproject.toml b/pyproject.toml index 024038e..98c1b6d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "agentcat" -version = "2.1.0" +version = "2.1.1" description = "Analytics tool for MCP (Model Context Protocol) servers, Claude Connectors, and ChatGPT Plugins - tracks tool usage patterns and provides insights" authors = [ { name = "AgentCat, Inc.", email = "support@agentcat.com" }, @@ -19,7 +19,7 @@ classifiers = [ ] dependencies = [ "mcp>=1.2.0,<3", - "agentcat-api==1.0.0", + "agentcat-api==1.0.2", "pydantic>=2.0.0,<3", "requests>=2.31.0", ] diff --git a/src/agentcat/modules/callpath.py b/src/agentcat/modules/callpath.py index 6d24eb8..71a2300 100644 --- a/src/agentcat/modules/callpath.py +++ b/src/agentcat/modules/callpath.py @@ -56,6 +56,8 @@ UserIdentity, ) +from .token_estimate import estimate_input_tokens, estimate_output_tokens + # A failed call whose adapter could make nothing of the failure. Same shape as # every other error payload so consumers never have to branch on presence. _UNKNOWN_ERROR: ErrorData = { @@ -305,6 +307,10 @@ async def publish_tool_call_event( is_error=is_error, error=(error or _UNKNOWN_ERROR) if is_error else None, duration=duration_ms, + # Estimated on the raw payloads here, before the queue's redaction + # hooks run; the queue never recomputes them. + input_tokens=estimate_input_tokens(raw_arguments), + output_tokens=estimate_output_tokens(response), client_name=rc.client.name, client_version=rc.client.version, identify_actor_given_id=rc.actor.user_id if rc.actor else None, diff --git a/src/agentcat/modules/token_estimate.py b/src/agentcat/modules/token_estimate.py new file mode 100644 index 0000000..72b81d2 --- /dev/null +++ b/src/agentcat/modules/token_estimate.py @@ -0,0 +1,112 @@ +"""Token estimates for tool-call events. + +One divisor, pinned byte-for-byte across the TypeScript, Python and Go SDKs: +``ceil(utf8_bytes / 3.5)``. Measured on production MCP responses, the inner +text runs 3.71 bytes per token on OpenAI tokenizers and about 3.15 on +Claude's; 3.5 splits the difference. The input side counts the raw arguments +the model emitted; the output side counts the content-block text the harness +feeds back. Nothing else counts. See the TypeScript repo's +docs/superpowers/specs/2026-09-19-sdk-token-estimates-design.md. +""" + +from __future__ import annotations + +import json +import math +from typing import Any + +TOKEN_ESTIMATE_BYTES_PER_TOKEN = 3.5 + +# The server's MAX_TOKEN_COUNT: the columns are int32. +_MAX_TOKEN_COUNT = 2_147_483_647 + + +def estimate_tokens(byte_count: int) -> int: + """``ceil(bytes / 3.5)``, 0 for nothing, clamped to the server's column.""" + if byte_count <= 0: + return 0 + return min( + math.ceil(byte_count / TOKEN_ESTIMATE_BYTES_PER_TOKEN), _MAX_TOKEN_COUNT + ) + + +def _utf8_len(text: str) -> int: + # surrogatepass: a lone surrogate off the wire (json.loads on a string + # containing an unpaired \udXXX escape) must still be counted, not + # dropped. TypeScript and Go's decoders substitute U+FFFD (3 bytes) for + # the same input; surrogatepass encodes a lone surrogate to the same 3 + # bytes, so the byte count matches across SDKs. + return len(text.encode("utf-8", errors="surrogatepass")) + + +def _compact_json_bytes(value: Any) -> int | None: + """UTF-8 length of the compact JSON: no spaces, no ASCII escaping. + + ``ensure_ascii=False`` matters: the default would spell ``café`` as + ``caf\\u00e9`` and count 20 bytes where TypeScript and Go count 16. No + ``default=str``: an unserializable value must be omitted, matching + TypeScript and Go, which also drop the field rather than counting a + stand-in string. + """ + try: + text = json.dumps(value, separators=(",", ":"), ensure_ascii=False) + except Exception: + return None + return _utf8_len(text) + + +def estimate_input_tokens(arguments: Any) -> int | None: + """Tokens the model spent emitting the call: the raw arguments, injected + parameters included. None when there are no arguments to count. + + Never raises: this runs inside the customer's request, and a failure to + estimate must cost at most this field, never the tool's response. + """ + try: + if arguments is None: + return None + byte_count = _compact_json_bytes(arguments) + return None if byte_count is None else estimate_tokens(byte_count) + except Exception: + return None + + +def estimate_output_tokens(response: Any) -> int | None: + """Tokens the model reads back: the text of the content blocks. + + Falls back to the whole response when it carries no content list; None + when there is no response at all. Never raises, for the same reason as + ``estimate_input_tokens``. + """ + try: + if response is None: + return None + content = response.get("content") if isinstance(response, dict) else None + if not isinstance(content, list): + byte_count = _compact_json_bytes(response) + return None if byte_count is None else estimate_tokens(byte_count) + if len(content) == 0: + structured = response.get("structuredContent") + if structured is None: + structured = response.get("structured_content") + if structured is not None: + byte_count = _compact_json_bytes(structured) + return None if byte_count is None else estimate_tokens(byte_count) + return 0 + return estimate_tokens( + sum(_content_block_bytes(block) for block in content) + ) + except Exception: + return None + + +def _content_block_bytes(block: Any) -> int: + if not isinstance(block, dict): + return 0 + if block.get("type") == "text" and isinstance(block.get("text"), str): + return _utf8_len(block["text"]) + if block.get("type") == "resource": + resource = block.get("resource") + if isinstance(resource, dict) and isinstance(resource.get("text"), str): + return _utf8_len(resource["text"]) + return 0 diff --git a/tests/test_token_estimate.py b/tests/test_token_estimate.py new file mode 100644 index 0000000..821cf0b --- /dev/null +++ b/tests/test_token_estimate.py @@ -0,0 +1,129 @@ +"""Shared token-estimate vectors, pinned byte-for-byte against the TypeScript +and Go SDKs. See the TypeScript repo's +docs/superpowers/specs/2026-09-19-sdk-token-estimates-design.md.""" + +import pytest + +from agentcat.modules.token_estimate import ( + TOKEN_ESTIMATE_BYTES_PER_TOKEN, + estimate_input_tokens, + estimate_output_tokens, + estimate_tokens, +) + + +def text(t: str) -> dict: + return {"type": "text", "text": t} + + +def test_the_divisor_is_pinned(): + assert TOKEN_ESTIMATE_BYTES_PER_TOKEN == 3.5 + + +@pytest.mark.parametrize( + "byte_count, tokens", + [(0, 0), (1, 1), (7, 2), (13, 4), (4096, 1171), (7516192765, 2147483647)], +) +def test_estimate_tokens(byte_count, tokens): + assert estimate_tokens(byte_count) == tokens + + +@pytest.mark.parametrize( + "arguments, tokens", + [ + ({"q": "hello"}, 4), + ({}, 1), + ({"name": "café"}, 5), # 20 bytes -> 6 under ensure_ascii; must be 16 -> 5 + ({"html": "&"}, 6), + ({"t": "你好"}, 4), + ({"ids": [1, 2, 3], "opts": {"deep": True, "n": None}}, 13), + ], +) +def test_estimate_input_tokens(arguments, tokens): + assert estimate_input_tokens(arguments) == tokens + + +def test_absent_arguments_are_omitted(): + assert estimate_input_tokens(None) is None + + +def test_lone_surrogate_counts_three_bytes_on_the_input_side(): + # A lone surrogate off the wire (json.loads on an unpaired \ud83d escape) + # must count, not be omitted: surrogatepass encodes it to 3 bytes, the + # same width Go counts after its decoder substitutes U+FFFD. + # {"a":""} = 6 + 3 + 2 = 11 bytes -> ceil(11 / 3.5) = 4. + assert estimate_input_tokens({"a": "\ud83d"}) == 4 + + +def test_unserializable_arguments_are_omitted(): + # No default=str: TypeScript and Go both omit an unserializable value + # rather than counting a stand-in string, so Python must match. + assert estimate_input_tokens({"when": object()}) is None + + +class _Hostile: + def __str__(self) -> str: + raise RuntimeError("boom") + + +class _HostileResponse(dict): + def get(self, *args, **kwargs): + raise RuntimeError("boom") + + +def test_never_raises_on_the_input_side(): + # json.dumps raises TypeError on an unserializable value before __str__ + # is ever consulted (no default=str); the field is omitted either way, + # so a hostile __str__ never gets the chance to raise into the call. + assert estimate_input_tokens({"when": _Hostile()}) is None + + +def test_never_raises_on_the_output_side(): + assert estimate_output_tokens(_HostileResponse(content=[])) is None + + +@pytest.mark.parametrize( + "response, tokens", + [ + ({"content": [text("hello world")]}, 4), + ({"content": [text("abcd"), text("e")]}, 2), + ({"content": [text("")]}, 0), + ({"content": [{"type": "image", "data": "QUJD", "mimeType": "image/png"}]}, 0), + ( + {"content": [{"type": "resource", "resource": {"uri": "file:///a", "text": "resource body"}}]}, + 4, + ), + ({"content": [{"type": "resource", "resource": {"uri": "file:///a", "blob": "QUJD"}}]}, 0), + ({"content": [text("hi")], "structuredContent": {"big": "y" * 1000}}, 1), + ({"content": [text("hi")], "structured_content": {"big": "y" * 1000}}, 1), + ({"content": [text("x" * 4096)]}, 1171), + ({"content": [text("hi")], "isError": True}, 1), + ({"content": [{"type": "text", "text": 42}, None, "str"]}, 0), + ({"content": [], "structuredContent": {"result": "ok"}}, 5), + ({"content": []}, 0), + ({"content": [], "structuredContent": None}, 0), + ( + { + "content": [{"type": "image", "data": "QUJD", "mimeType": "image/png"}], + "structuredContent": {"result": "ok"}, + }, + 0, + ), + ({"content": [], "structured_content": {"result": "ok"}}, 5), + ], +) +def test_estimate_output_tokens(response, tokens): + assert estimate_output_tokens(response) == tokens + + +def test_no_content_list_falls_back_to_the_whole_response(): + assert estimate_output_tokens({"result": "ok"}) == 5 + + +def test_lone_surrogate_counts_three_bytes_on_the_output_side(): + # A lone surrogate is 3 bytes -> ceil(3 / 3.5) = 1. + assert estimate_output_tokens({"content": [text("\ud83d")]}) == 1 + + +def test_absent_response_is_omitted(): + assert estimate_output_tokens(None) is None diff --git a/tests/test_token_estimates_integration.py b/tests/test_token_estimates_integration.py new file mode 100644 index 0000000..a2cca9c --- /dev/null +++ b/tests/test_token_estimates_integration.py @@ -0,0 +1,234 @@ +"""Token estimates ride on every tools/call event, computed on the raw +payloads before the queue's redaction and truncation stages run. + +Vectors: {"text":"hi there"} is 19 bytes -> 6 tokens; the flavors' `echo` +tool answers "echo:hi there", 13 bytes -> 4. {"text":"secret-value-123456789"} +is 33 bytes -> 10, and the echo of it, "echo:secret-value-123456789", is 27 +bytes -> 8. +""" + +import pytest + +from agentcat import AgentCatOptions, track +from agentcat.modules import event_queue + +from .test_utils.flavors import flavors + + +@pytest.fixture(autouse=True) +def capture(monkeypatch): + """Collect every event the queue is handed, without touching the network.""" + events: list = [] + monkeypatch.setattr(event_queue.event_queue, "add", events.append) + return events + + +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_tool_call_events_carry_token_estimates(flavor, capture): + built = flavor.build("token-estimates") + track(built.server, "proj_test", AgentCatOptions()) + + async with flavor.client(built.server) as client: + await flavor.list_tools(client) + await flavor.call(client, "echo", {"text": "hi there"}) + + event = capture[0] + assert event.input_tokens == 6 + # Only the content text counts: the structured mirror of the same answer + # and the era-specific error/structured keys add nothing. + assert event.output_tokens == 4 + + +def _build_structured_only_server(flavor_id: str): + """A fresh, untracked server whose lone tool answers with an empty + ``content`` list and a ``structuredContent``/``structured_content`` of + ``{"result": "ok"}`` — no auto-mirrored text block. + + Each era's facade normally derives a text content block from a typed + return value, so this bypasses that conversion the way the era itself + allows: ``MCPServer`` and the community ``fastmcp`` both pass a + already-built result object straight through their `convert_result` + (`mcp.server.mcpserver.utilities.func_metadata.FuncMetadata.convert_result`, + the community `Tool.convert_result`) instead of re-deriving content + from it, and the lowlevel `Server`'s `on_call_tool` callback is returned + to the wire completely unmodified. + """ + if flavor_id == "mcpserver-v2": + from mcp.server.mcpserver import MCPServer + from mcp.types import CallToolResult + + server = MCPServer("structured-only") + + @server.tool() + async def structured_only() -> CallToolResult: + return CallToolResult(content=[], structured_content={"result": "ok"}) + + return server + + if flavor_id == "lowlevel-v2": + from mcp import types + from mcp.server import Server + + async def on_list_tools(ctx, params): + return types.ListToolsResult( + tools=[ + types.Tool( + name="structured_only", + description="", + input_schema={"type": "object", "properties": {}}, + ) + ] + ) + + async def on_call_tool(ctx, params): + return types.CallToolResult(content=[], structured_content={"result": "ok"}) + + return Server( + "structured-only", on_list_tools=on_list_tools, on_call_tool=on_call_tool + ) + + if flavor_id.startswith("community-"): + from fastmcp import FastMCP + + # `fastmcp.tools` re-exports ToolResult in every supported release; + # the module behind it moved (`fastmcp.tools.tool` through 3.1.x, + # `fastmcp.tools.base` from 3.2), so import from the package. + from fastmcp.tools import ToolResult + + server = FastMCP("structured-only") + + @server.tool + async def structured_only() -> ToolResult: + return ToolResult(content=[], structured_content={"result": "ok"}) + + return server + + return None + + +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_structured_only_result_counts_the_structured_content(flavor, capture): + """When a result's `content` list is empty and it carries a structured + value, `output_tokens` counts that value's compact JSON instead of 0. + + Not every flavor's facade can be made to answer with an empty `content` + list without going around its typed-return conversion; flavors that + cannot are skipped here rather than faked, per the design brief. + """ + server = _build_structured_only_server(flavor.id) + if server is None: + pytest.skip( + f"{flavor.id}: no known way to make this flavor answer with an " + "empty content list and a structured value" + ) + track(server, "proj_test", AgentCatOptions()) + + async with flavor.client(server) as client: + await flavor.call(client, "structured_only", {}) + + event = capture[0] + # Confirms the fixture itself, not just the estimate: the event records + # the customer's undecorated result, an empty content list with the + # structured value intact — the wire response the client actually sees + # also carries the SDK's session mint-back text, which must not count. + assert event.response["content"] == [] + structured = event.response.get("structured_content") or event.response.get( + "structuredContent" + ) + assert structured == {"result": "ok"} + assert event.output_tokens == 5 # {"result":"ok"} = 15 bytes + + +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_counts_survive_redaction_and_truncation(flavor, capture, monkeypatch): + sent: list = [] + monkeypatch.setattr(event_queue.event_queue, "_send_event", sent.append) + + built = flavor.build("token-estimates-redacted") + track( + built.server, + "proj_test", + # Only the secret string is rewritten: a hook that replaced every + # string would also clobber the content block's `type` discriminator. + AgentCatOptions( + redact_sensitive_information=lambda text: ( + "[REDACTED]" if "secret" in text else text + ) + ), + ) + + async with flavor.client(built.server) as client: + await flavor.list_tools(client) + await flavor.call(client, "echo", {"text": "secret-value-123456789"}) + + queued = capture[0] + assert queued.redaction_fn is not None + # Drive the real pipeline (redact -> sanitize -> truncate -> send) on the + # captured event, exactly as the worker thread would. + event_queue.event_queue._process_event(queued) + + assert len(sent) == 1 + published = sent[0] + assert published.parameters["arguments"]["text"] == "[REDACTED]" + assert published.response["content"][0]["text"] == "[REDACTED]" + assert published.input_tokens == 10 + assert published.output_tokens == 8 # "echo:secret-value-123456789" = 27 bytes + + +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_counts_survive_truncation(flavor, capture, monkeypatch): + """{"text":"xxx...x"} (60000 x's) is 60011 bytes -> ceil(60011/3.5) = 17146 + exactly. The echo of it, "echo:" + 60000 x's, is 60005 bytes -> + ceil(60005/3.5) = 17145. The brief's original 40000-x vector (11432 / + 11430) does not reliably push every flavor's serialized event past + truncation.MAX_EVENT_BYTES (100KB) — some flavors mirror the answer into + `structuredContent` too and some do not, so only the larger payload + guarantees size-targeted truncation fires on every flavor. The published + response text ends up well below the original 60005 characters, and the + counts still describe the original, untruncated bytes. + """ + sent: list = [] + monkeypatch.setattr(event_queue.event_queue, "_send_event", sent.append) + + built = flavor.build("token-estimates-truncated") + track(built.server, "proj_test", AgentCatOptions()) + + long_text = "x" * 60000 + async with flavor.client(built.server) as client: + await flavor.list_tools(client) + await flavor.call(client, "echo", {"text": long_text}) + + queued = capture[0] + # Drive the real pipeline (sanitize -> truncate -> send) on the captured + # event, exactly as the worker thread would. + event_queue.event_queue._process_event(queued) + + assert len(sent) == 1 + published = sent[0] + published_text = published.response["content"][0]["text"] + assert len(published_text) < 60005 + assert published.input_tokens == 17146 + assert published.output_tokens == 17145 + + +@pytest.mark.parametrize("flavor", flavors(), ids=lambda f: f.id) +async def test_a_broken_estimator_never_breaks_the_tool_call(flavor, capture, monkeypatch): + """The estimator runs inside the customer's request. Force it to raise + and the client must still receive the tool's own answer; the call's + analytics may be lost, the response never is.""" + + def explode(_value): + raise RuntimeError("estimator bug") + + monkeypatch.setattr("agentcat.modules.callpath.estimate_input_tokens", explode) + monkeypatch.setattr("agentcat.modules.callpath.estimate_output_tokens", explode) + + built = flavor.build("token-estimates-broken") + track(built.server, "proj_test", AgentCatOptions()) + + async with flavor.client(built.server) as client: + await flavor.list_tools(client) + called = await flavor.call(client, "echo", {"text": "hi there"}) + + assert not called.is_error + assert "echo:hi there" in called.text