diff --git a/CHANGELOG.md b/CHANGELOG.md index 59d0ef2..2ef6838 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,6 +20,9 @@ and versions are tracked in the repo-root `VERSION` file. - Align the Typer support floor with the tested matrix and cover representative minimum/maximum Typer and Click version pairings. +- Prefix formula-leading CSV/TSV cells with an apostrophe by default to protect + spreadsheet consumers; pass `formula_guard=False` only for an audited raw + value contract. ### Fixed diff --git a/docs/api-reference.md b/docs/api-reference.md index 56abbd1..94e0384 100644 --- a/docs/api-reference.md +++ b/docs/api-reference.md @@ -1318,7 +1318,7 @@ base_cli.option(...) ### `render_document` **Kind:** function -**Signature:** `render_document(document: 'Mapping[str, Any]', *, requested_format: 'str | None', records_key: 'str | None' = None, columns: 'Sequence[tuple[str, str]] | None' = None, stream: 'TextIO | None' = None) -> 'str'` +**Signature:** `render_document(document: 'Mapping[str, Any]', *, requested_format: 'str | None', records_key: 'str | None' = None, columns: 'Sequence[tuple[str, str]] | None' = None, stream: 'TextIO | None' = None, formula_guard: 'bool' = True) -> 'str'` **Behavior:** Render a structured report or leave terminal text to its existing renderer. @@ -1334,7 +1334,7 @@ base_cli.render_document(...) ### `render_records` **Kind:** function -**Signature:** `render_records(records: 'Iterable[Mapping[str, Any]]', *, requested_format: 'str | None', columns: 'Sequence[tuple[str, str]]', stream: 'TextIO | None' = None, footer: 'str | None' = None, minimum_widths: 'Sequence[int] | None' = None, terminal_width: 'int | None' = None, max_cell_width: 'int | None' = 80, rich: 'bool' = False) -> 'str'` +**Signature:** `render_records(records: 'Iterable[Mapping[str, Any]]', *, requested_format: 'str | None', columns: 'Sequence[tuple[str, str]]', stream: 'TextIO | None' = None, footer: 'str | None' = None, minimum_widths: 'Sequence[int] | None' = None, terminal_width: 'int | None' = None, max_cell_width: 'int | None' = 80, rich: 'bool' = False, formula_guard: 'bool' = True) -> 'str'` **Behavior:** Render records according to the shared public output contract. diff --git a/docs/output-contracts.md b/docs/output-contracts.md index 5fe5fd2..fe6423e 100644 --- a/docs/output-contracts.md +++ b/docs/output-contracts.md @@ -16,6 +16,11 @@ Delimited output is intentionally automation-friendly: - no column header or footer is emitted; - values use the standard `csv` quoting rules, while ANSI escape sequences and other control characters are replaced with spaces. +- cells beginning with `=`, `+`, `-`, `@`, tab, or carriage return receive a + leading apostrophe by default so spreadsheet programs treat them as text + rather than formulas; + pass `formula_guard=False` only when a downstream consumer explicitly needs + the original leading character. `ndjson` is the bounded machine-output format for large or long-running results. It consumes the input iterable once and writes one flushed JSON object diff --git a/lib/python/base_cli/output.py b/lib/python/base_cli/output.py index 4258c41..a8ad397 100644 --- a/lib/python/base_cli/output.py +++ b/lib/python/base_cli/output.py @@ -23,6 +23,7 @@ _ANSI_ESCAPE_RE = re.compile(r"\x1b(?:\[[0-?]*[ -/]*[@-~]|\][^\x07]*(?:\x07|\x1b\\))") _DEFAULT_TERMINAL_WIDTH = 120 _DEFAULT_MAX_CELL_WIDTH = 80 +_FORMULA_TRIGGER_CHARS = frozenset("=+-@\t\r") class OutputFormatError(ValueError): @@ -121,6 +122,7 @@ def render_records( terminal_width: int | None = None, max_cell_width: int | None = _DEFAULT_MAX_CELL_WIDTH, rich: bool = False, + formula_guard: bool = True, ) -> str: """Render records according to the shared public output contract. @@ -132,7 +134,10 @@ def render_records( columns. Terminal cells use Unicode display-cell widths and are bounded by ``terminal_width`` and ``max_cell_width`` with deterministic ellipsis truncation. ``rich=True`` opts terminal text into the optional Rich - renderer and otherwise falls back to the built-in table. + renderer and otherwise falls back to the built-in table. ``formula_guard`` + controls the default apostrophe prefix for formula-leading CSV/TSV cells; + disabling it is an explicit security decision for consumers that need raw + values. """ target = stream if stream is not None else sys.stdout @@ -144,7 +149,7 @@ def render_records( delimiter = "," if resolved == "csv" else "\t" writer = csv.writer(target, delimiter=delimiter, lineterminator="\n") for row in record_list: - writer.writerow([_delimited_value(row.get(key)) for _header, key in columns]) + writer.writerow([_delimited_value(row.get(key), formula_guard=formula_guard) for _header, key in columns]) return resolved if resolved == "ndjson": @@ -187,13 +192,16 @@ def render_document( records_key: str | None = None, columns: Sequence[tuple[str, str]] | None = None, stream: TextIO | None = None, + formula_guard: bool = True, ) -> str: """Render a structured report or leave terminal text to its existing renderer. Structured formats preserve the complete document. Delimited output uses the selected record list (or the document itself) and never emits report - prose, headers, or footers. A terminal ``text`` request returns ``text`` - without writing so the caller can keep its established human report. + prose, headers, or footers. ``formula_guard`` has the same CSV/TSV security + behavior as ``render_records``. A terminal ``text`` request returns + ``text`` without writing so the caller can keep its established human + report. """ target = stream if stream is not None else sys.stdout @@ -233,6 +241,7 @@ def render_document( requested_format=resolved, columns=selected_columns, stream=target, + formula_guard=formula_guard, ) return resolved @@ -272,7 +281,7 @@ def _validate_delimited_records( dumps_strict_json(value, separators=(",", ":")) -def _delimited_value(value: Any) -> str: +def _delimited_value(value: Any, *, formula_guard: bool = True) -> str: """Return a safe scalar for redirected CSV/TSV output. Delimited output is commonly piped into another process. Keep the normal @@ -281,7 +290,11 @@ def _delimited_value(value: Any) -> str: record across physical lines. """ - return _table_cell(_cell_value(value)) + raw_cell = _cell_value(value) + cell = _table_cell(raw_cell) + if formula_guard and raw_cell[:1] in _FORMULA_TRIGGER_CHARS: + return f"'{cell}" + return cell def _write_table( diff --git a/tests/test_output.py b/tests/test_output.py index 15bf6a3..0f39f9c 100644 --- a/tests/test_output.py +++ b/tests/test_output.py @@ -12,6 +12,7 @@ NDJSON_SCHEMA_VERSION, NdjsonWriter, OutputFormatError, + _delimited_value, render_document, render_records, resolve_output_format, @@ -91,6 +92,48 @@ def test_delimited_emitters_validate_nested_values_before_writing(self) -> None: ) self.assertEqual(stream.getvalue(), "") + def test_delimited_emitters_guard_spreadsheet_formulas_by_default(self) -> None: + records = ({"name": "=SUM(A1:A2)", "path": "+cmd"}, {"name": "-10", "path": "@user"}) + + for requested_format, expected in ( + ("csv", "'=SUM(A1:A2),'+cmd\n'-10,'@user\n"), + ("tsv", "'=SUM(A1:A2)\t'+cmd\n'-10\t'@user\n"), + ): + with self.subTest(format=requested_format): + stream = io.StringIO() + render_records(records, requested_format=requested_format, columns=COLUMNS, stream=stream) + self.assertEqual(stream.getvalue(), expected) + + def test_delimited_formula_guard_can_be_disabled_explicitly(self) -> None: + stream = io.StringIO() + + render_records( + ({"name": "=SUM(A1:A2)", "path": "@user"},), + requested_format="csv", + columns=COLUMNS, + stream=stream, + formula_guard=False, + ) + + self.assertEqual(stream.getvalue(), "=SUM(A1:A2),@user\n") + + def test_delimited_formula_guard_covers_tab_and_carriage_return_before_sanitizing(self) -> None: + values = ("\t=SUM(A1:A2)", "\r@user") + + with mock.patch("base_cli.output._table_cell", side_effect=lambda value: value): + guarded = [_delimited_value(value) for value in values] + + self.assertEqual(guarded, ["'\t=SUM(A1:A2)", "'\r@user"]) + + stream = io.StringIO() + render_records( + ({"name": values[0], "path": values[1]},), + requested_format="csv", + columns=COLUMNS, + stream=stream, + ) + self.assertEqual(next(csv.reader(io.StringIO(stream.getvalue()))), ["' =SUM(A1:A2)", "' @user"]) + def test_tsv_consumes_one_pass_iterable_without_materializing(self) -> None: consumed = False