Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions pyproject.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
[project]
name = "agentcat"
version = "2.1.0"
version = "2.1.1"
description = "Analytics tool for MCP (Model Context Protocol) servers, Claude Connectors, and ChatGPT Plugins - tracks tool usage patterns and provides insights"
authors = [
{ name = "AgentCat, Inc.", email = "support@agentcat.com" },
Expand All @@ -19,7 +19,7 @@ classifiers = [
]
dependencies = [
"mcp>=1.2.0,<3",
"agentcat-api==1.0.0",
"agentcat-api==1.0.2",
"pydantic>=2.0.0,<3",
"requests>=2.31.0",
]
Expand Down
6 changes: 6 additions & 0 deletions src/agentcat/modules/callpath.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,8 @@
UserIdentity,
)

from .token_estimate import estimate_input_tokens, estimate_output_tokens

# A failed call whose adapter could make nothing of the failure. Same shape as
# every other error payload so consumers never have to branch on presence.
_UNKNOWN_ERROR: ErrorData = {
Expand Down Expand Up @@ -305,6 +307,10 @@ async def publish_tool_call_event(
is_error=is_error,
error=(error or _UNKNOWN_ERROR) if is_error else None,
duration=duration_ms,
# Estimated on the raw payloads here, before the queue's redaction
# hooks run; the queue never recomputes them.
input_tokens=estimate_input_tokens(raw_arguments),
output_tokens=estimate_output_tokens(response),
client_name=rc.client.name,
client_version=rc.client.version,
identify_actor_given_id=rc.actor.user_id if rc.actor else None,
Expand Down
112 changes: 112 additions & 0 deletions src/agentcat/modules/token_estimate.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,112 @@
"""Token estimates for tool-call events.

One divisor, pinned byte-for-byte across the TypeScript, Python and Go SDKs:
``ceil(utf8_bytes / 3.5)``. Measured on production MCP responses, the inner
text runs 3.71 bytes per token on OpenAI tokenizers and about 3.15 on
Claude's; 3.5 splits the difference. The input side counts the raw arguments
the model emitted; the output side counts the content-block text the harness
feeds back. Nothing else counts. See the TypeScript repo's
docs/superpowers/specs/2026-09-19-sdk-token-estimates-design.md.
"""

from __future__ import annotations

import json
import math
from typing import Any

TOKEN_ESTIMATE_BYTES_PER_TOKEN = 3.5

# The server's MAX_TOKEN_COUNT: the columns are int32.
_MAX_TOKEN_COUNT = 2_147_483_647


def estimate_tokens(byte_count: int) -> int:
"""``ceil(bytes / 3.5)``, 0 for nothing, clamped to the server's column."""
if byte_count <= 0:
return 0
return min(
math.ceil(byte_count / TOKEN_ESTIMATE_BYTES_PER_TOKEN), _MAX_TOKEN_COUNT
)


def _utf8_len(text: str) -> int:
# surrogatepass: a lone surrogate off the wire (json.loads on a string
# containing an unpaired \udXXX escape) must still be counted, not
# dropped. TypeScript and Go's decoders substitute U+FFFD (3 bytes) for
# the same input; surrogatepass encodes a lone surrogate to the same 3
# bytes, so the byte count matches across SDKs.
return len(text.encode("utf-8", errors="surrogatepass"))


def _compact_json_bytes(value: Any) -> int | None:
"""UTF-8 length of the compact JSON: no spaces, no ASCII escaping.

``ensure_ascii=False`` matters: the default would spell ``café`` as
``caf\\u00e9`` and count 20 bytes where TypeScript and Go count 16. No
``default=str``: an unserializable value must be omitted, matching
TypeScript and Go, which also drop the field rather than counting a
stand-in string.
"""
try:
text = json.dumps(value, separators=(",", ":"), ensure_ascii=False)
except Exception:
return None
return _utf8_len(text)


def estimate_input_tokens(arguments: Any) -> int | None:
"""Tokens the model spent emitting the call: the raw arguments, injected
parameters included. None when there are no arguments to count.

Never raises: this runs inside the customer's request, and a failure to
estimate must cost at most this field, never the tool's response.
"""
try:
if arguments is None:
return None
byte_count = _compact_json_bytes(arguments)
return None if byte_count is None else estimate_tokens(byte_count)
except Exception:
return None


def estimate_output_tokens(response: Any) -> int | None:
"""Tokens the model reads back: the text of the content blocks.

Falls back to the whole response when it carries no content list; None
when there is no response at all. Never raises, for the same reason as
``estimate_input_tokens``.
"""
try:
if response is None:
return None
content = response.get("content") if isinstance(response, dict) else None
if not isinstance(content, list):
byte_count = _compact_json_bytes(response)
return None if byte_count is None else estimate_tokens(byte_count)
if len(content) == 0:
structured = response.get("structuredContent")
if structured is None:
structured = response.get("structured_content")
if structured is not None:
byte_count = _compact_json_bytes(structured)
return None if byte_count is None else estimate_tokens(byte_count)
return 0
return estimate_tokens(
sum(_content_block_bytes(block) for block in content)
)
except Exception:
return None


def _content_block_bytes(block: Any) -> int:
if not isinstance(block, dict):
return 0
if block.get("type") == "text" and isinstance(block.get("text"), str):
return _utf8_len(block["text"])
if block.get("type") == "resource":
resource = block.get("resource")
if isinstance(resource, dict) and isinstance(resource.get("text"), str):
return _utf8_len(resource["text"])
return 0
129 changes: 129 additions & 0 deletions tests/test_token_estimate.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,129 @@
"""Shared token-estimate vectors, pinned byte-for-byte against the TypeScript
and Go SDKs. See the TypeScript repo's
docs/superpowers/specs/2026-09-19-sdk-token-estimates-design.md."""

import pytest

from agentcat.modules.token_estimate import (
TOKEN_ESTIMATE_BYTES_PER_TOKEN,
estimate_input_tokens,
estimate_output_tokens,
estimate_tokens,
)


def text(t: str) -> dict:
return {"type": "text", "text": t}


def test_the_divisor_is_pinned():
assert TOKEN_ESTIMATE_BYTES_PER_TOKEN == 3.5


@pytest.mark.parametrize(
"byte_count, tokens",
[(0, 0), (1, 1), (7, 2), (13, 4), (4096, 1171), (7516192765, 2147483647)],
)
def test_estimate_tokens(byte_count, tokens):
assert estimate_tokens(byte_count) == tokens


@pytest.mark.parametrize(
"arguments, tokens",
[
({"q": "hello"}, 4),
({}, 1),
({"name": "café"}, 5), # 20 bytes -> 6 under ensure_ascii; must be 16 -> 5
({"html": "<a>&</a>"}, 6),
({"t": "你好"}, 4),
({"ids": [1, 2, 3], "opts": {"deep": True, "n": None}}, 13),
],
)
def test_estimate_input_tokens(arguments, tokens):
assert estimate_input_tokens(arguments) == tokens


def test_absent_arguments_are_omitted():
assert estimate_input_tokens(None) is None


def test_lone_surrogate_counts_three_bytes_on_the_input_side():
# A lone surrogate off the wire (json.loads on an unpaired \ud83d escape)
# must count, not be omitted: surrogatepass encodes it to 3 bytes, the
# same width Go counts after its decoder substitutes U+FFFD.
# {"a":"<surrogate>"} = 6 + 3 + 2 = 11 bytes -> ceil(11 / 3.5) = 4.
assert estimate_input_tokens({"a": "\ud83d"}) == 4


def test_unserializable_arguments_are_omitted():
# No default=str: TypeScript and Go both omit an unserializable value
# rather than counting a stand-in string, so Python must match.
assert estimate_input_tokens({"when": object()}) is None


class _Hostile:
def __str__(self) -> str:
raise RuntimeError("boom")


class _HostileResponse(dict):
def get(self, *args, **kwargs):
raise RuntimeError("boom")


def test_never_raises_on_the_input_side():
# json.dumps raises TypeError on an unserializable value before __str__
# is ever consulted (no default=str); the field is omitted either way,
# so a hostile __str__ never gets the chance to raise into the call.
assert estimate_input_tokens({"when": _Hostile()}) is None


def test_never_raises_on_the_output_side():
assert estimate_output_tokens(_HostileResponse(content=[])) is None


@pytest.mark.parametrize(
"response, tokens",
[
({"content": [text("hello world")]}, 4),
({"content": [text("abcd"), text("e")]}, 2),
({"content": [text("")]}, 0),
({"content": [{"type": "image", "data": "QUJD", "mimeType": "image/png"}]}, 0),
(
{"content": [{"type": "resource", "resource": {"uri": "file:///a", "text": "resource body"}}]},
4,
),
({"content": [{"type": "resource", "resource": {"uri": "file:///a", "blob": "QUJD"}}]}, 0),
({"content": [text("hi")], "structuredContent": {"big": "y" * 1000}}, 1),
({"content": [text("hi")], "structured_content": {"big": "y" * 1000}}, 1),
({"content": [text("x" * 4096)]}, 1171),
({"content": [text("hi")], "isError": True}, 1),
({"content": [{"type": "text", "text": 42}, None, "str"]}, 0),
({"content": [], "structuredContent": {"result": "ok"}}, 5),
({"content": []}, 0),
({"content": [], "structuredContent": None}, 0),
(
{
"content": [{"type": "image", "data": "QUJD", "mimeType": "image/png"}],
"structuredContent": {"result": "ok"},
},
0,
),
({"content": [], "structured_content": {"result": "ok"}}, 5),
],
)
def test_estimate_output_tokens(response, tokens):
assert estimate_output_tokens(response) == tokens


def test_no_content_list_falls_back_to_the_whole_response():
assert estimate_output_tokens({"result": "ok"}) == 5


def test_lone_surrogate_counts_three_bytes_on_the_output_side():
# A lone surrogate is 3 bytes -> ceil(3 / 3.5) = 1.
assert estimate_output_tokens({"content": [text("\ud83d")]}) == 1


def test_absent_response_is_omitted():
assert estimate_output_tokens(None) is None
Loading
Loading