From c1a9dcf805f02c7e723b3c6b4a34e4c988a36a60 Mon Sep 17 00:00:00 2001 From: Ramesh Padmanabhaiah <22363102+codeforester@users.noreply.github.com> Date: Thu, 8 Oct 2026 10:58:14 +0530 Subject: [PATCH 1/2] fix: stream delimited output without materializing records --- docs/output-contracts.md | 3 +++ lib/python/base_cli/output.py | 26 +++++++++----------------- tests/test_output.py | 17 +++++++++++++++-- 3 files changed, 27 insertions(+), 19 deletions(-) diff --git a/docs/output-contracts.md b/docs/output-contracts.md index fe6423e..286d83d 100644 --- a/docs/output-contracts.md +++ b/docs/output-contracts.md @@ -12,6 +12,9 @@ Delimited output is intentionally automation-friendly: - rows are streamed directly from the iterable, so CSV and TSV do not retain the complete result set in memory; +- each row is fully serialized and validated before it is written. If a later + row cannot be encoded, earlier rows remain in the sink and the renderer + raises the encoding error; - the supplied `columns` sequence controls both column order and cell lookup; - no column header or footer is emitted; - values use the standard `csv` quoting rules, while ANSI escape sequences and diff --git a/lib/python/base_cli/output.py b/lib/python/base_cli/output.py index e33cdf1..418dc29 100644 --- a/lib/python/base_cli/output.py +++ b/lib/python/base_cli/output.py @@ -144,12 +144,17 @@ def render_records( resolved = resolve_output_format(requested_format, stream=target) if resolved in ("csv", "tsv"): - record_list = [dict(record) for record in records] - _validate_delimited_records(record_list, columns) delimiter = "," if resolved == "csv" else "\t" writer = csv.writer(target, delimiter=delimiter, lineterminator="\n") - for row in record_list: - writer.writerow([_delimited_value(row.get(key), formula_guard=formula_guard) for _header, key in columns]) + for record in records: + # Serialize and validate a complete row before touching the sink. + # This keeps each row atomic while allowing earlier rows to flow + # through for one-pass and unbounded producers. + row = [ + _delimited_value(record.get(key), formula_guard=formula_guard) + for _header, key in columns + ] + writer.writerow(row) return resolved if resolved == "ndjson": @@ -268,19 +273,6 @@ def _cell_value(value: Any) -> str: return str(value) -def _validate_delimited_records( - records: Sequence[Mapping[str, Any]], - columns: Sequence[tuple[str, str]], -) -> None: - """Validate nested cell values before a delimited stream is touched.""" - - for record in records: - for _header, key in columns: - value = record.get(key) - if isinstance(value, (Mapping, list, tuple)): - dumps_strict_json(value, separators=(",", ":")) - - def _delimited_value(value: Any, *, formula_guard: bool = True) -> str: """Return a safe scalar for redirected CSV/TSV output. diff --git a/tests/test_output.py b/tests/test_output.py index c0bb726..20b8473 100644 --- a/tests/test_output.py +++ b/tests/test_output.py @@ -82,7 +82,7 @@ def test_json_emitters_reject_non_finite_values_without_partial_output(self) -> emit(stream) self.assertEqual(stream.getvalue(), "") - def test_delimited_emitters_validate_nested_values_before_writing(self) -> None: + def test_delimited_emitters_validate_each_row_before_writing_it(self) -> None: records = ({"name": "valid"}, {"name": {"value": float("nan")}}) for requested_format in ("csv", "tsv"): with self.subTest(format=requested_format): @@ -91,7 +91,20 @@ def test_delimited_emitters_validate_nested_values_before_writing(self) -> None: render_records( records, requested_format=requested_format, columns=(("NAME", "name"),), stream=stream ) - self.assertEqual(stream.getvalue(), "") + self.assertEqual(stream.getvalue(), "valid\n") + + def test_delimited_emitters_request_the_next_record_only_after_writing_previous_row(self) -> None: + stream = io.StringIO() + + def records(): + yield {"name": "first"} + assert stream.getvalue() == "first\n" + yield {"name": {"value": float("nan")}} + + with self.assertRaises(ValueError): + render_records(records(), requested_format="tsv", columns=(("NAME", "name"),), stream=stream) + + self.assertEqual(stream.getvalue(), "first\n") def test_delimited_emitters_guard_spreadsheet_formulas_by_default(self) -> None: records = ({"name": "=SUM(A1:A2)", "path": "+cmd"}, {"name": "-10", "path": "@user"}) From bc2a4290c84f31de52627cd56ff702d8b92e6802 Mon Sep 17 00:00:00 2001 From: Ramesh Padmanabhaiah <22363102+codeforester@users.noreply.github.com> Date: Thu, 8 Oct 2026 22:32:29 +0530 Subject: [PATCH 2/2] docs: clarify streaming delimited output --- CHANGELOG.md | 2 ++ docs/json-contracts.md | 5 +++++ lib/python/base_cli/output.py | 5 +---- 3 files changed, 8 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 6b4e752..cdf531e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -34,6 +34,8 @@ and versions are tracked in the repo-root `VERSION` file. - Prefix formula-leading CSV/TSV cells with an apostrophe by default to protect spreadsheet consumers; pass `formula_guard=False` only for an audited raw value contract. +- Stream CSV/TSV rows with bounded memory and document that earlier rows remain + available when a later row cannot be encoded. - Add the namespaced `BASE_CLI_LOG_UTC` environment variable and deprecate `LOG_UTC` with a migration warning; the legacy alias is scheduled for removal diff --git a/docs/json-contracts.md b/docs/json-contracts.md index 996fb99..0635f21 100644 --- a/docs/json-contracts.md +++ b/docs/json-contracts.md @@ -18,6 +18,11 @@ app = base_cli.App( envelope on stdout. Logs remain on stderr. Human mode, including the default Click error rendering and command stdout behavior, is unchanged. +JSON output remains all-or-nothing at that envelope boundary: a failed command +emits an error envelope rather than exposing a partially emitted JSON document. +CSV and TSV are intentionally different streaming formats; if a later row +cannot be encoded, rows already written to the sink remain available. + Capture is activated only when JSON mode is selected, including an environment variable or Click `default_map`. Human and NDJSON invocations write directly to the caller's stdout, preserving progress visibility and flush behavior. JSON diff --git a/lib/python/base_cli/output.py b/lib/python/base_cli/output.py index 418dc29..e7303bc 100644 --- a/lib/python/base_cli/output.py +++ b/lib/python/base_cli/output.py @@ -150,10 +150,7 @@ def render_records( # Serialize and validate a complete row before touching the sink. # This keeps each row atomic while allowing earlier rows to flow # through for one-pass and unbounded producers. - row = [ - _delimited_value(record.get(key), formula_guard=formula_guard) - for _header, key in columns - ] + row = [_delimited_value(record.get(key), formula_guard=formula_guard) for _header, key in columns] writer.writerow(row) return resolved