Skip to content

Commit b5a8a59

Browse files
committed
implement: Convert a solutions file on its own when no questions file is beside it (t30)
2 parents 6c705a4 + a666d79 commit b5a8a59

10 files changed

Lines changed: 229 additions & 36 deletions

File tree

‎docs/how-it-works.md‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -384,7 +384,7 @@ these 21, in this order:
384384
| --- | --- |
385385
| `source` | the document, relative to the corpus root |
386386
| `set` | the folder the document is in |
387-
| `outcome` | `built`, `build refused`, `faulted`, `skipped`, `no spec`, `no model`, `spec rejected`, `bad spec`, or `error: <exception>` |
387+
| `outcome` | `built`, `build refused`, `faulted`, `skipped`, `no spec`, `no model`, `spec failed` and `fix failed` where the model call did not finish, `spec rejected`, `bad spec`, or `error: <exception>` |
388388
| `reason` | the build's refusal, the first error the checks still found, the warnings a build proceeded past, or what an exception said |
389389
| `spec` | `wrote`, `reused`, or `rewritten` where the spec rewrite ran |
390390
| `layout` | the coverage's layout |

‎in2lambda_agent/cli.py‎

Lines changed: 5 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -9,7 +9,7 @@
99

1010
from in2lambda_agent import compare, corpus, gate, pipeline
1111
from in2lambda_agent.mathpix import MathpixClient, MathpixError
12-
from in2lambda_agent.model import ModelUnavailable, choose_backend
12+
from in2lambda_agent.model import ModelError, ModelUnavailable, choose_backend
1313
from in2lambda_agent.ocr import ocr_pdf
1414
from in2lambda_agent.package import CommandRefused, SpecRejected
1515
from in2lambda_agent.review import ReviewError
@@ -443,14 +443,16 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
443443
except (
444444
MathpixError,
445445
ModelUnavailable,
446+
ModelError,
446447
BadSpec,
447448
SpecRejected,
448449
ReviewError,
449450
CommandRefused,
450451
) as error:
451452
# Missing credentials among them: the message names the variables, or
452-
# the login to run, or what a spec says that a spec cannot say, or the
453-
# question a review command names that is not under review.
453+
# the login to run, or what the provider said stopped a call, or what a
454+
# spec says that a spec cannot say, or the question a review command
455+
# names that is not under review.
454456
print(f"in2lambda-agent: {error}", file=sys.stderr)
455457
return 1
456458

‎in2lambda_agent/corpus.py‎

Lines changed: 8 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -25,7 +25,7 @@
2525
from typing import Optional, Sequence
2626

2727
from in2lambda_agent import package, pair, pipeline
28-
from in2lambda_agent.model import Backend, ModelUnavailable
28+
from in2lambda_agent.model import Backend, ModelError, ModelUnavailable
2929
from in2lambda_agent.package import SpecRejected, is_document
3030
from in2lambda_agent.settings import Settings
3131
from in2lambda_agent.spec import RECORD_NAME, SPEC_NAME, BadSpec
@@ -62,7 +62,8 @@ class Row:
6262
the checks still fault and no zip, `skipped` for a file that is not
6363
a document and for a solutions document with no questions document
6464
beside it, `no spec` for a replay with nothing saved to replay,
65-
`no model`, `spec rejected`, `bad spec`, or `error: <exception>`.
65+
`no model`, `spec failed` and `fix failed` where a model call did
66+
not finish, `spec rejected`, `bad spec`, or `error: <exception>`.
6667
reason: What the run had to say for itself, in the words of whatever
6768
said it: the refusal, the first error the checks were still finding,
6869
or what the exception said. On a `built` row it holds the warnings
@@ -287,6 +288,11 @@ def run_one(
287288
except ModelUnavailable as error:
288289
row.outcome = "no model" if existed else "no spec"
289290
row.reason = _one_line(str(error))
291+
except ModelError as error:
292+
# Which call did not finish, and what the provider said it stopped on.
293+
# A row reading `error: ResultError` says neither.
294+
row.outcome = f"{error.stage or 'model'} failed"
295+
row.reason = _one_line(str(error))
290296
except SpecRejected as error:
291297
row.outcome = "spec rejected"
292298
row.reason = _one_line(str(error))

‎in2lambda_agent/model.py‎

Lines changed: 66 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -14,6 +14,9 @@
1414
1515
Every `Reply` carries the tokens and the wall time for that call, which the
1616
design spec's test plan records per document.
17+
18+
A call that does not finish raises `ModelError`, whose message is what the
19+
provider said. A call that cannot be made at all raises `ModelUnavailable`.
1720
"""
1821

1922
import asyncio
@@ -46,6 +49,35 @@
4649

4750
REQUEST_TIMEOUT = 300.0
4851

52+
# Claude Code's own tools, named for `disallowed_tools`. The agent gives the
53+
# model the tools each call needs and no others: a spec call has none, and a
54+
# call that could run Bash or Read on the paths its prompt names spends its
55+
# turns reading the corpus. `tools=[]` alone does not switch them off — the SDK
56+
# sends it as `--tools ""`, and the run that recorded `error_max_turns` on
57+
# Worksheet_1.pdf passed it. `disallowed_tools` refuses each tool by name, and
58+
# a name Claude Code does not have is ignored.
59+
BUILTIN_TOOLS = (
60+
"Agent",
61+
"Bash",
62+
"BashOutput",
63+
"Edit",
64+
"ExitPlanMode",
65+
"Glob",
66+
"Grep",
67+
"KillShell",
68+
"LS",
69+
"MultiEdit",
70+
"NotebookEdit",
71+
"Read",
72+
"Skill",
73+
"SlashCommand",
74+
"Task",
75+
"TodoWrite",
76+
"WebFetch",
77+
"WebSearch",
78+
"Write",
79+
)
80+
4981

5082
@dataclass
5183
class Tool:
@@ -99,6 +131,19 @@ class ModelUnavailable(RuntimeError):
99131
"""A backend was called without the credential or the login it needs."""
100132

101133

134+
class ModelError(RuntimeError):
135+
"""A call was made and did not finish: the provider stopped it, or the
136+
model asked for tools until the round limit and never answered.
137+
138+
Attributes:
139+
stage: Which of the agent's calls this was — `spec` or `fix` — set by
140+
the pipeline and read by the corpus sweep, which names it in the
141+
row's outcome. Empty where nothing set it.
142+
"""
143+
144+
stage: str = ""
145+
146+
102147
def _encoded(image: bytes) -> str:
103148
"""One PNG page as the base64 every provider's image block carries."""
104149
return base64.standard_b64encode(image).decode("ascii")
@@ -137,6 +182,7 @@ def call(
137182
138183
Raises:
139184
ModelUnavailable: If `unavailable` would give a reason.
185+
ModelError: If the call did not finish.
140186
"""
141187

142188

@@ -182,6 +228,7 @@ async def _call(
182228
) -> Reply:
183229
from claude_agent_sdk import (
184230
ClaudeAgentOptions,
231+
ClaudeSDKError,
185232
ResultMessage,
186233
create_sdk_mcp_server,
187234
query,
@@ -202,14 +249,20 @@ async def handler(arguments: dict[str, Any]) -> dict[str, Any]:
202249
options = ClaudeAgentOptions(
203250
system_prompt=system,
204251
mcp_servers={"agent": server},
252+
# The call's own tools, and no others: `allowed_tools` is empty for
253+
# the spec call, which has none.
205254
allowed_tools=[f"mcp__agent__{one.name}" for one in tools],
255+
disallowed_tools=list(BUILTIN_TOOLS),
206256
# No built-in tools, and no settings file: nothing the machine
207257
# happens to have configured reaches the call. Both need the empty
208258
# list, which the SDK documents as "disable all built-in tools" and
209259
# "disable filesystem settings"; the default for each is `None`,
210260
# which loads the CLI's own set.
211261
tools=[],
212262
setting_sources=[],
263+
# No permission prompt: a call has no terminal to answer one at,
264+
# and the tools it may run are the two lists above.
265+
permission_mode="bypassPermissions",
213266
max_turns=MAX_TOOL_ROUNDS,
214267
)
215268

@@ -249,17 +302,26 @@ async def one_message():
249302

250303
result = None
251304
stream = query(prompt=asked, options=options)
305+
failed = None
252306
try:
253307
async for message in stream:
254308
if isinstance(message, ResultMessage) and result is None:
255309
result = message
310+
except ClaudeSDKError as error:
311+
# The SDK raises rather than yielding a result for a run the CLI
312+
# ended on an error, so the two branches below never see one. Its
313+
# message says what stopped the run; raising it here rather than
314+
# inside the `async for` keeps the `finally` below.
315+
failed = error
256316
finally:
257317
await stream.aclose()
258318

319+
if failed is not None:
320+
raise ModelError(str(failed))
259321
if result is None:
260-
raise RuntimeError("the agent-sdk backend returned no result")
322+
raise ModelError("the agent-sdk backend returned no result")
261323
if result.is_error:
262-
raise RuntimeError(
324+
raise ModelError(
263325
f"the agent-sdk backend stopped on {result.subtype}: "
264326
f"{result.result}"
265327
)
@@ -389,7 +451,7 @@ def call(
389451
)
390452
messages.append({"role": "user", "content": results})
391453

392-
raise RuntimeError(
454+
raise ModelError(
393455
f"the {self.name} backend asked for tools for "
394456
f"{MAX_TOOL_ROUNDS} rounds without answering"
395457
)
@@ -535,7 +597,7 @@ def _loop(
535597
}
536598
)
537599

538-
raise RuntimeError(
600+
raise ModelError(
539601
f"the {self.name} backend asked for tools for "
540602
f"{MAX_TOOL_ROUNDS} rounds without answering"
541603
)

‎in2lambda_agent/pipeline.py‎

Lines changed: 45 additions & 22 deletions
Original file line numberDiff line numberDiff line change
@@ -36,7 +36,13 @@
3636
from in2lambda_agent import package, pair
3737
from in2lambda_agent.fix import RoundResult, fix_round, summary, unrepaired
3838
from in2lambda_agent.mathpix import MathpixClient
39-
from in2lambda_agent.model import Backend, ModelUnavailable, Usage, choose_backend
39+
from in2lambda_agent.model import (
40+
Backend,
41+
ModelError,
42+
ModelUnavailable,
43+
Usage,
44+
choose_backend,
45+
)
4046
from in2lambda_agent.ocr import MEDIA_NAME, cached, ocr_pdf
4147
from in2lambda_agent.review import RECORD, Question, Review, choose
4248
from in2lambda_agent.settings import Settings
@@ -176,6 +182,8 @@ def run(
176182
them when a conversion is needed and the run has no Mathpix
177183
credentials. A PDF already in the cache needs none.
178184
ModelUnavailable: If a spec must be written and no backend can run.
185+
ModelError: If a call did not finish, with `stage` naming which — the
186+
spec call or a fixing round.
179187
BadSpec: If what the model answers with is not a spec.
180188
SpecRejected: If in2lambda will not run the spec.
181189
SourceError: If in2lambda cannot freeze or check the source.
@@ -274,18 +282,24 @@ def run(
274282
if (reason := backend.unavailable()) is not None:
275283
raise ModelUnavailable(reason)
276284
result.second = _second(source, cache_dir, solutions)
277-
draft, coverage, report, result.tries = iterate_spec(
278-
frozen,
279-
saved,
280-
backend,
281-
tries=tries,
282-
on_stage=result.add_stage,
283-
second=result.second,
284-
previous=previous,
285-
solutions=frozen_solutions,
286-
solutions_name=solutions.name if solutions is not None else "",
287-
solutions_only=alone is not None,
288-
)
285+
try:
286+
draft, coverage, report, result.tries = iterate_spec(
287+
frozen,
288+
saved,
289+
backend,
290+
tries=tries,
291+
on_stage=result.add_stage,
292+
second=result.second,
293+
previous=previous,
294+
solutions=frozen_solutions,
295+
solutions_name=solutions.name if solutions is not None else "",
296+
solutions_only=alone is not None,
297+
)
298+
except ModelError as error:
299+
# Which call did not finish, for a caller that names it: a spec call
300+
# and a fixing round both go to the same backend.
301+
error.stage = "spec"
302+
raise
289303
result.draft = draft
290304
result.coverage = coverage
291305
for one in result.tries:
@@ -295,7 +309,11 @@ def run(
295309

296310
# Layers 3 and 4, a round at a time. Reached only with a spec this run
297311
# wrote, so the backend is the one that wrote it.
298-
report = _fix_rounds(draft, report, backend, rounds, result)
312+
try:
313+
report = _fix_rounds(draft, report, backend, rounds, result)
314+
except ModelError as error:
315+
error.stage = "fix"
316+
raise
299317
# What the corpus harness reads off the result rather than off the
300318
# record: set here so that a run that stops for a review carries them
301319
# too, since that return is above the record this run never writes.
@@ -406,6 +424,7 @@ def resume(
406424
Raises:
407425
ReviewError: no review is waiting, or none of its questions is `key`.
408426
ModelUnavailable: a rejection has no backend to answer its note with.
427+
ModelError: a rejection's fixing round did not finish.
409428
CommandRefused: in2lambda would not make the reviewer's edit.
410429
"""
411430
cache_dir = Path(cache_dir).resolve()
@@ -488,14 +507,18 @@ def resume(
488507
raise ModelUnavailable(reason)
489508
# The note is a finding of its own: the checks are quiet, and it is
490509
# what the round is for. Rounds after it answer what they leave.
491-
report = _fix_rounds(
492-
draft,
493-
package.validate(draft),
494-
backend,
495-
waiting.limit,
496-
result,
497-
instruction=f"The reviewer rejected {key}: {note}",
498-
)
510+
try:
511+
report = _fix_rounds(
512+
draft,
513+
package.validate(draft),
514+
backend,
515+
waiting.limit,
516+
result,
517+
instruction=f"The reviewer rejected {key}: {note}",
518+
)
519+
except ModelError as error:
520+
error.stage = "fix"
521+
raise
499522
relisted = [key]
500523
else:
501524
package.command(

‎in2lambda_agent/spec.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -365,6 +365,7 @@ def iterate_spec(
365365
prints went to `on_stage` as the loop made it.
366366
367367
Raises:
368+
ModelError: a call did not finish.
368369
BadSpec: what the model answered with is not a spec.
369370
SpecRejected: in2lambda will not run a spec this loop wrote.
370371
SourceError: in2lambda cannot freeze or check this source.

‎in2lambda_agent/ui/server.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -37,7 +37,7 @@
3737

3838
from in2lambda_agent import corpus, pipeline, spec
3939
from in2lambda_agent.mathpix import MathpixError
40-
from in2lambda_agent.model import ModelUnavailable
40+
from in2lambda_agent.model import ModelError, ModelUnavailable
4141
from in2lambda_agent.package import CommandRefused, SpecRejected
4242
from in2lambda_agent.review import RECORD, ReviewError
4343
from in2lambda_agent.settings import Settings, load_settings
@@ -62,6 +62,7 @@
6262
FAILURES = (
6363
MathpixError,
6464
ModelUnavailable,
65+
ModelError,
6566
BadSpec,
6667
SpecRejected,
6768
ReviewError,

‎tests/conftest.py‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -18,7 +18,8 @@ class FakeBackend:
1818
A reply is the text to answer with, or a list of `(tool name, arguments)`
1919
for a call that uses its tools: the named tools are run in the order given,
2020
against whatever they were built over, exactly as a real backend's loop runs
21-
them. That is what scripts a fixing round without a model in it.
21+
them. That is what scripts a fixing round without a model in it. A reply
22+
that is an exception is raised, which scripts a call that does not finish.
2223
"""
2324

2425
name = "fake"
@@ -36,6 +37,8 @@ def call(self, system, prompt, tools=(), images=()):
3637
self.calls.append((system, prompt))
3738
self.images.append(list(images))
3839
reply = self.replies.pop(0)
40+
if isinstance(reply, Exception):
41+
raise reply
3942
made = []
4043
if isinstance(reply, list):
4144
by_name = {one.name: one for one in tools}

0 commit comments

Comments
 (0)