Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 8 additions & 1 deletion configs/pi/extensions/workflow-context.ts
Original file line number Diff line number Diff line change
Expand Up @@ -44,9 +44,16 @@ function runInBackground(args: string[], cwd?: string): void {
export default function workflowContextExtension(pi: ExtensionAPI): void {
pi.on("before_agent_start", async (event: any, ctx: any) => {
try {
// 45s, matching the Claude hook `witan setup` installs. 5s cleared
// the warm output-cache hit (0.6-0.9s measured in agent-kit#349)
// but sat far below the cold path (16-23s there), so every prompt
// that missed the 30s cache silently contributed no block. A
// timeout is still wanted — a hung read must degrade to no
// context rather than stall the turn — but it has to sit above
// the cold path, not inside it.
const r = spawnSync("witan", ["inject-context"], {
encoding: "utf8",
timeout: 5000,
timeout: 45000,
cwd: ctx?.cwd,
});
const text = (r.stdout ?? "").trim();
Expand Down
34 changes: 34 additions & 0 deletions mcp/servers/witan/CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,40 @@ a MINOR bump may include breaking changes).

## [Unreleased]

### Fixed

- **`task_ready` no longer re-reads blockers it has already fetched.** Its
repo-scoped branch scans every Task and then narrows to the candidates, but
resolved blocker statuses against the narrowed set — so every blocker living
in another repo was unknown and fetched again, one `get_task` per blocker.
Measured against a deployment that was ~3.0s of a 4.2s call, on identical
scans to a 1.2s `task_list`.

- **The context hook issues its independent reads together.** `witan
inject-context` against a deployment made up to ten sequential tool calls, so
the cold path was their sum. It now runs in two waves (the second needs the
first's answers), which measured 11.3s -> 6.5s on the same graph, byte-identical
output. Each read keeps its own failure isolation, and a machine that cannot
start threads falls back to running them in line. The wave primes the proxy's
tool schema once first, so the fan-out does not have every worker discover it
(4 `tools/list` cold, against 1 for the sequential path it replaced); against
a `witan-core` predating `prime_tool_schema` this is skipped and the old
behaviour stands.

- **The context hook's ready list respects cross-repo blockers.** It offers a
repo-scoped slice of an all-Task scan, and an unresolvable blocker counts as
closed — so a task blocked by an open task in another repo was advertised as
ready to work. `readiness.filter_ready` now takes the wider row set, which
the hook already had in hand.

- **The prompt and stop hooks get timeouts above what they are timing.** Both
were 15s. `witan inject-context` was measured at 16-23s cold on a large graph,
so the hook was killed mid-read and the user paid the full wait for no block;
it is now 45s. `witan session-checkpoint` writes, and a write has been measured
at up to 51s, so it is now 60s — being killed there leaves a session open with
no handoff summary. The pi `workflow-context` extension was tighter still at
5s and now matches at 45s.

## [0.36.0] - 2026-09-18

### Added
Expand Down
173 changes: 173 additions & 0 deletions mcp/servers/witan/tests/test_context.py
Original file line number Diff line number Diff line change
Expand Up @@ -1336,3 +1336,176 @@ def test_inject_context_remote_asks_for_held_tasks_across_all_repos(
"task_list",
{"assignee": "@me", "status": "in_progress", "repo": ""},
) in server.calls


def test_inject_context_remote_issues_its_independent_reads_concurrently(
tmp_path, monkeypatch
):
"""The cold path is the slowest read, not the sum of them.

Issued one at a time, the hook's ten round trips against a deployment took
~11s measured and blew the 15s timeout it installs (agent-kit#349). Each
read here blocks on a barrier that only releases once every first-wave
call has arrived, so a sequential implementation deadlocks and times out
rather than quietly passing slower.
"""
import threading

from witan import context as ctx_module

monkeypatch.setenv("TMPDIR", str(tmp_path))
monkeypatch.setenv("WITAN_CONTEXT_TTL", "0")
import tempfile

monkeypatch.setattr(tempfile, "tempdir", None)
repo = "https://github.com/test/ctx-concurrent"
monkeypatch.setenv("WITAN_REPO", repo)
monkeypatch.setattr(ctx_module, "_current_branch", lambda: "main")

# The four first-wave reads. The second wave (sessions, comments) depends
# on their answers, so it is deliberately not held here.
first_wave = {
"workflow_project_list",
"task_ready",
"task_for_branch",
"task_list",
}
barrier = threading.Barrier(len(first_wave), timeout=10)

class _Barriered(_FakeRemoteServer):
def _guard(self, name):
if name in first_wave:
barrier.wait()
super()._guard(name)

server = _Barriered(
projects=[{"slug": "wp-x", "title": "X", "phase": "spec"}],
ready=[],
sessions_by_project={},
branch_tasks=[],
held=[],
)
text = ctx_module.inject_context_remote(
server, "https://witan.example.org/mcp", debug=True
)

assert barrier.broken is False
assert "## Active Workflow Projects" in text


def test_inject_context_remote_primes_the_tool_schema_once_before_fanning_out(
tmp_path, monkeypatch
):
"""One schema resolution per wave, and only when the proxy offers one.

Without it each worker finds the param-name cache unset and lists the
surface itself (measured: 4 `tools/list` cold, where the sequential path
issued 1). The catch-all `__getattr__` case is the trap: a proxy predating
`prime_tool_schema` must be detected as NOT having it, rather than having
a bogus tool of that name called on the deployment.
"""
from witan import context as ctx_module

monkeypatch.setenv("TMPDIR", str(tmp_path))
monkeypatch.setenv("WITAN_CONTEXT_TTL", "0")
import tempfile

monkeypatch.setattr(tempfile, "tempdir", None)
monkeypatch.setenv("WITAN_REPO", "https://github.com/test/ctx-prime")

class _Priming(_FakeRemoteServer):
primed = 0

def prime_tool_schema(self):
type(self).primed += 1

server = _Priming(projects=[], ready=[], sessions_by_project={})
ctx_module.inject_context_remote(server, "https://witan.example.org/mcp")
assert _Priming.primed == 1
assert not [c for c in server.calls if c[0] == "prime_tool_schema"]

# A proxy-shaped object whose __getattr__ answers for ANY name, i.e. one
# from a witan-core that predates the method.
class _CatchAll(_FakeRemoteServer):
def __getattr__(self, name):
def _tool_call(**kwargs):
self.calls.append((name, kwargs))
raise RuntimeError(f"no such tool: {name}")

return _tool_call

legacy = _CatchAll(projects=[], ready=[], sessions_by_project={})
ctx_module.inject_context_remote(legacy, "https://witan.example.org/mcp")
assert not [c for c in legacy.calls if c[0] == "prime_tool_schema"]


def test_gather_falls_back_to_serial_when_the_pool_cannot_start(monkeypatch):
"""A machine at its thread limit still gets its answers, just slower.

``_gather`` is called outside the remote path's own try/except, so raising
here would crash a hook whose entire contract is that it degrades quietly.
"""
from witan import context as ctx_module

def _no_threads(*_args, **_kwargs):
raise RuntimeError("can't start new thread")

monkeypatch.setattr(ctx_module, "ThreadPoolExecutor", _no_threads)

def _boom():
raise ValueError("read failed")

got = ctx_module._gather({"a": lambda: 1, "b": lambda: 2, "c": _boom})

assert got["a"] == 1
assert got["b"] == 2
# And a genuine read failure is still reported as a value, not raised.
assert isinstance(got["c"], ValueError)


def test_gather_reports_each_failure_against_its_own_key():
from witan import context as ctx_module

def _boom():
raise ValueError("nope")

got = ctx_module._gather({"ok": lambda: "fine", "bad": _boom})

assert got["ok"] == "fine"
assert isinstance(got["bad"], ValueError)


def test_inject_context_remote_one_failed_read_costs_only_its_own_block(
tmp_path, monkeypatch
):
"""Per-read isolation must survive the move into worker threads.

A read that fails inside a worker comes back as a value rather than a live
exception, so the block it feeds is the only thing that may disappear.
"""
from witan import context as ctx_module

monkeypatch.setenv("TMPDIR", str(tmp_path))
monkeypatch.setenv("WITAN_CONTEXT_TTL", "0")
import tempfile

monkeypatch.setattr(tempfile, "tempdir", None)
repo = "https://github.com/test/ctx-isolated"
monkeypatch.setenv("WITAN_REPO", repo)
monkeypatch.setattr(ctx_module, "_current_branch", lambda: "main")

server = _FakeRemoteServer(
projects=[{"slug": "wp-x", "title": "X", "phase": "spec"}],
ready=[{"slug": "tk-r", "title": "Ready one", "priority": "p1"}],
sessions_by_project={},
branch_tasks=[{"slug": "tk-b", "title": "Branch one", "status": "open"}],
raises=("task_for_branch",),
)
text = ctx_module.inject_context_remote(
server, "https://witan.example.org/mcp", debug=True
)

assert "## In-Flight Branch" not in text
assert "## Active Workflow Projects" in text
assert "## Ready Tasks" in text
assert "Ready one" in text
38 changes: 38 additions & 0 deletions mcp/servers/witan/tests/test_readiness.py
Original file line number Diff line number Diff line change
Expand Up @@ -128,3 +128,41 @@ def test_filter_ready_orders_by_priority_and_reclaims_expired():
def test_filter_ready_unknown_blocker_treated_closed():
tasks = [{"slug": "tk-x", "status": "open", "blocked_by": ["tk-gone"]}]
assert [t["slug"] for t in readiness.filter_ready(tasks)] == ["tk-x"]


def test_filter_ready_resolves_blockers_from_the_wider_row_set():
"""A blocker outside the candidate slice holds its task back.

The context hook offers a repo-scoped slice of an all-Task scan. Without
the wider set the open blocker below is simply unknown, and unknown reads
as closed — so the task would be advertised as ready to work while the
thing blocking it is wide open (agent-kit#349).
"""
tasks = [{"slug": "tk-x", "status": "open", "blocked_by": ["tk-other-repo"]}]
others = [
{"slug": "tk-other-repo", "status": "open", "repo": "https://example.com/b"}
]

assert [t["slug"] for t in readiness.filter_ready(tasks)] == ["tk-x"]
assert readiness.filter_ready(tasks, blocker_rows=others) == []

others[0]["status"] = "closed"
assert [t["slug"] for t in readiness.filter_ready(tasks, blocker_rows=others)] == [
"tk-x"
]


def test_filter_ready_candidate_status_wins_over_a_stale_wider_row():
"""Where a slug is in both sets, the candidate row decides.

``tk-b`` is closed among the candidates and open in the wider set. Reading
the wider copy would hold ``tk-a`` back against data the caller has already
superseded. (``tk-b`` itself is absent either way — closed is not pickable.)
"""
tasks = [
{"slug": "tk-a", "status": "open", "blocked_by": ["tk-b"]},
{"slug": "tk-b", "status": "closed"},
]
stale = [{"slug": "tk-b", "status": "open"}]
got = [t["slug"] for t in readiness.filter_ready(tasks, blocker_rows=stale)]
assert got == ["tk-a"]
9 changes: 7 additions & 2 deletions mcp/servers/witan/tests/test_setup.py
Original file line number Diff line number Diff line change
Expand Up @@ -48,8 +48,13 @@ def test_witan_bundle_registers_witan_mcp_server_and_hooks(tmp_path, monkeypatch
]
assert inject and checkpoint
# Both prompt-path hooks carry a timeout so a hung git/graph can't stall.
assert inject[0]["timeout"] == 15
assert checkpoint[0]["timeout"] == 15
# It has to sit ABOVE the cold-path read cost, not inside it: at 15s the
# hook was killed mid-read on a large graph, so the user paid the full
# wait and got no block (agent-kit#349).
assert inject[0]["timeout"] == 45
# Longer than the read hook: this one is a write, and a write against a
# deployment has been measured at up to 51s.
assert checkpoint[0]["timeout"] == 60


def test_mcp_only_platforms_keep_the_self_contained_uvx_entry(tmp_path, monkeypatch):
Expand Down
43 changes: 43 additions & 0 deletions mcp/servers/witan/tests/test_tasks.py
Original file line number Diff line number Diff line change
Expand Up @@ -217,6 +217,49 @@ def test_create_with_already_closed_blocker_is_open(server):
assert b["slug"] in {t["slug"] for t in server.task_ready()}


@requires_omnigraph
def test_ready_resolves_a_cross_repo_blocker_without_re_reading_it(server, monkeypatch):
"""A blocker in another repo is already in the all-Task scan — use it.

``task_ready``'s repo-scoped branch reads every Task and then narrows to
the candidates. Resolving blocker statuses against the NARROW set left
every cross-repo blocker unknown, so each one was fetched again one
``get_task`` at a time: ~3.0s of a 4.2s call against the deployed service
and most of what blew the context hook's 15s timeout (agent-kit#349).
"""
from witan import server as srv

other = "https://github.com/test/other"
blocker = server.task_create(title="blocker elsewhere", description="x", repo=other)
dependent = server.task_create(
title="dependent here", description="x", blocked_by=[blocker["slug"]]
)

reads: list[str] = []
real_read = srv.client.read

def counting_read(queries, name, params):
reads.append(name)
return real_read(queries, name, params)

monkeypatch.setattr(srv.client, "read", counting_read)
ready = {t["slug"] for t in server.task_ready(repo="https://github.com/test/repo")}

# The open blocker lives in another repo, so it is not a candidate — but it
# still holds its dependent back.
assert blocker["slug"] not in ready
assert dependent["slug"] not in ready
assert "get_task" not in reads

# And the answer is not merely "everything is excluded": closing the
# blocker releases the dependent, still off the same two scans.
server.task_close(blocker["slug"])
reads.clear() # after the close, whose own reads are not what is counted
ready = {t["slug"] for t in server.task_ready(repo="https://github.com/test/repo")}
assert dependent["slug"] in ready
assert "get_task" not in reads


@requires_omnigraph
def test_update_to_closed_unblocks_dependents(server):
a = server.task_create(title="blocker", description="x")
Expand Down
Loading
Loading