Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,8 @@ and versions are tracked in the repo-root `VERSION` file.
- Prefix formula-leading CSV/TSV cells with an apostrophe by default to protect
spreadsheet consumers; pass `formula_guard=False` only for an audited raw
value contract.
- Stream CSV/TSV rows with bounded memory and document that earlier rows remain
available when a later row cannot be encoded.

- Add the namespaced `BASE_CLI_LOG_UTC` environment variable and deprecate
`LOG_UTC` with a migration warning; the legacy alias is scheduled for removal
Expand Down
5 changes: 5 additions & 0 deletions docs/json-contracts.md
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,11 @@ app = base_cli.App(
envelope on stdout. Logs remain on stderr. Human mode, including the default
Click error rendering and command stdout behavior, is unchanged.

JSON output remains all-or-nothing at that envelope boundary: a failed command
emits an error envelope rather than exposing a partially emitted JSON document.
CSV and TSV are intentionally different streaming formats; if a later row
cannot be encoded, rows already written to the sink remain available.

Capture is activated only when JSON mode is selected, including an environment
variable or Click `default_map`. Human and NDJSON invocations write directly to
the caller's stdout, preserving progress visibility and flush behavior. JSON
Expand Down
3 changes: 3 additions & 0 deletions docs/output-contracts.md
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,9 @@ Delimited output is intentionally automation-friendly:

- rows are streamed directly from the iterable, so CSV and TSV do not retain
the complete result set in memory;
- each row is fully serialized and validated before it is written. If a later
row cannot be encoded, earlier rows remain in the sink and the renderer
raises the encoding error;
- the supplied `columns` sequence controls both column order and cell lookup;
- no column header or footer is emitted;
- values use the standard `csv` quoting rules, while ANSI escape sequences and
Expand Down
23 changes: 6 additions & 17 deletions lib/python/base_cli/output.py
Original file line number Diff line number Diff line change
Expand Up @@ -144,12 +144,14 @@ def render_records(
resolved = resolve_output_format(requested_format, stream=target)

if resolved in ("csv", "tsv"):
record_list = [dict(record) for record in records]
_validate_delimited_records(record_list, columns)
delimiter = "," if resolved == "csv" else "\t"
writer = csv.writer(target, delimiter=delimiter, lineterminator="\n")
for row in record_list:
writer.writerow([_delimited_value(row.get(key), formula_guard=formula_guard) for _header, key in columns])
for record in records:
# Serialize and validate a complete row before touching the sink.
# This keeps each row atomic while allowing earlier rows to flow
# through for one-pass and unbounded producers.
row = [_delimited_value(record.get(key), formula_guard=formula_guard) for _header, key in columns]
writer.writerow(row)
return resolved

if resolved == "ndjson":
Expand Down Expand Up @@ -268,19 +270,6 @@ def _cell_value(value: Any) -> str:
return str(value)


def _validate_delimited_records(
records: Sequence[Mapping[str, Any]],
columns: Sequence[tuple[str, str]],
) -> None:
"""Validate nested cell values before a delimited stream is touched."""

for record in records:
for _header, key in columns:
value = record.get(key)
if isinstance(value, (Mapping, list, tuple)):
dumps_strict_json(value, separators=(",", ":"))


def _delimited_value(value: Any, *, formula_guard: bool = True) -> str:
"""Return a safe scalar for redirected CSV/TSV output.

Expand Down
17 changes: 15 additions & 2 deletions tests/test_output.py
Original file line number Diff line number Diff line change
Expand Up @@ -82,7 +82,7 @@ def test_json_emitters_reject_non_finite_values_without_partial_output(self) ->
emit(stream)
self.assertEqual(stream.getvalue(), "")

def test_delimited_emitters_validate_nested_values_before_writing(self) -> None:
def test_delimited_emitters_validate_each_row_before_writing_it(self) -> None:
records = ({"name": "valid"}, {"name": {"value": float("nan")}})
for requested_format in ("csv", "tsv"):
with self.subTest(format=requested_format):
Expand All @@ -91,7 +91,20 @@ def test_delimited_emitters_validate_nested_values_before_writing(self) -> None:
render_records(
records, requested_format=requested_format, columns=(("NAME", "name"),), stream=stream
)
self.assertEqual(stream.getvalue(), "")
self.assertEqual(stream.getvalue(), "valid\n")

def test_delimited_emitters_request_the_next_record_only_after_writing_previous_row(self) -> None:
stream = io.StringIO()

def records():
yield {"name": "first"}
assert stream.getvalue() == "first\n"
yield {"name": {"value": float("nan")}}

with self.assertRaises(ValueError):
render_records(records(), requested_format="tsv", columns=(("NAME", "name"),), stream=stream)

self.assertEqual(stream.getvalue(), "first\n")

def test_delimited_emitters_guard_spreadsheet_formulas_by_default(self) -> None:
records = ({"name": "=SUM(A1:A2)", "path": "+cmd"}, {"name": "-10", "path": "@user"})
Expand Down
Loading