diff --git a/.github/workflows/repo-description-drift.yml b/.github/workflows/repo-description-drift.yml new file mode 100644 index 0000000..ebad022 --- /dev/null +++ b/.github/workflows/repo-description-drift.yml @@ -0,0 +1,41 @@ +name: repository-description-drift + +on: + push: + branches: [main] + paths: + - facts/facts.json + - facts/facts.schema.json + - facts/repo-descriptions.json + - scripts/repo_descriptions.py + - .github/workflows/repo-description-drift.yml + workflow_dispatch: + +permissions: + contents: read + +jobs: + public-descriptions: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + - name: Render exact proposed descriptions and review-only commands + run: | + mkdir -p description-proposals + python3 scripts/repo_descriptions.py render --facts facts/facts.json > description-proposals/descriptions.json + python3 scripts/repo_descriptions.py render --facts facts/facts.json --format commands > description-proposals/apply-commands.txt + - name: Check live public descriptions without applying changes + env: + GH_TOKEN: ${{ github.token }} + run: python3 scripts/repo_descriptions.py check --facts facts/facts.json --live > description-proposals/live-check.json + - name: Preserve exact proposals and live deltas even on drift + if: always() + uses: actions/upload-artifact@v4 + with: + name: repository-description-proposals + path: description-proposals/ + if-no-files-found: error diff --git a/facts/README.md b/facts/README.md index 4ca7709..23fa7b7 100644 --- a/facts/README.md +++ b/facts/README.md @@ -54,3 +54,76 @@ format separators are escaped by the renderer so GitHub keeps every column. The hub's `publish` workflow triggers only on `v*` tags; `handle-upstream` only on upstream release dispatches. A facts-only branch or main change triggers neither. The ordinary test workflow may still run. + +## Public repository descriptions + +`repo-descriptions.json` is the versioned description policy, separate from the +registry-derived facts. It contains only confirmed-public repository identities, +approved product wording, templates, and license labels with exact public +`LICENSE` commit URLs. It makes no new licensing decision. Held repositories +retain their published text, are checked for drift, and never receive a command. +Nonpublic repositories are excluded from public policy, output, and artifacts. + +Quantities, dates, release identifiers, and product commits use +`{{facts:token.name|format}}` from the explicitly supplied `facts.json`; formats +are `int`, `grouped` (thousands separators), `decimal`, and `text` (strings only). +For example, `releases.1.0.1.scoreboard.graded` selects that immutable release, +not current main. `{{license:id}}` inserts a centrally approved label and records +its pinned source. `{{word:id}}` inserts approved static product wording, such as +language/ABI standard names; it is not a quantity store. + +The engine preserves the approved published stopgap exactly. Its parenthesized +open-corpus strategy and closed-test script counts are **historical inventory** +from `inventory`, not the current graded-probe denominator. The policy records +that scope explicitly. Corpus copy names the historical source release separately +from current graded probes. Advancing the scoreboard never refills inventory or +changes a released scoreboard. + +The approved engine all-graded claim fails closed unless `scoreboard.belowStrong` +is zero and `scoreboard.excellent + scoreboard.strong == scoreboard.graded`. +An unsupported snapshot is refused before output or live API calls, so it cannot +produce partial command proposals. GitHub owner/repository identities compare +case-insensitively for source verification, duplicates, live reads, and holds; +pinned commits and the `LICENSE` source path still match exactly. + +At **every promotion**, without delayed batching: + +1. Immediately export canonical facts using `lab facts export`, following the + existing exporter/pinned-inventory process. A pipeline never creates a model. +2. Render repository, README, website, and hosted-consumer outputs from that same + explicit snapshot; refresh pinned consumer copies and run `lab facts render` + and `lab facts check` for marked documentation. +3. Render and inspect exact descriptions, source tokens, license sources, hashes, + and the proposed command text: + + ```sh + python3 scripts/repo_descriptions.py render --facts facts/facts.json > descriptions.json + python3 scripts/repo_descriptions.py render --facts facts/facts.json --format commands > apply-commands.txt + ``` + +4. TOP alone reviews and applies applicable commands. The renderer does not + execute its output; there is no apply mode. Held rows emit no command. +5. Prove exact live agreement immediately after application: + + ```sh + python3 scripts/repo_descriptions.py check --facts facts/facts.json --live + ``` + +Use `--policy FILE` for a reviewed alternate policy. `check --gh PATH` accepts an +operator-supplied executable without embedding workstation paths in source. +Checks issue one read-only GitHub API GET per public row with a sixty-second +timeout, no shell, and exact public identity verification. JSON output includes +expected/actual deltas; null is distinct from empty text. Exit codes are zero for +all matches, one for drift (including held text), and two for input, API, auth, or +timeout errors. Failed reads never count as matches. Descriptions must be +nonempty, at most 350 Unicode characters, and free of line separators/control +characters; malformed JSON, duplicate keys, unknown fields/tokens, invalid +numbers, and unresolved templates fail closed. Facts are validated against the +unchanged public schema with the supported standard-library validator. + +The read-only `repository-description-drift` workflow runs on relevant pushes to +main and manual dispatch, not pull requests. It fails on drift/errors and uploads +exact proposals, review-only command text, and live deltas even on failure. It +has no write permission or automatic application. Offline CLI acceptance uses a +fake third-party executable, not live GitHub; run it in the approved test +environment with `python3 -m unittest discover -s tests -p 'test_repo_descriptions.py' -v`. diff --git a/facts/repo-descriptions.json b/facts/repo-descriptions.json new file mode 100644 index 0000000..2c9a682 --- /dev/null +++ b/facts/repo-descriptions.json @@ -0,0 +1,113 @@ +{ + "schema": "pineforge/repo-descriptions/v1", + "version": 1, + "labels": { + "apache": "Apache-2.0", + "source": "PineForge Source License 1.2" + }, + "licenses": { + "engine": { + "label": "apache", + "repo": "pineforge-4pass/pineforge-engine", + "commit": "59082e696f0c7a95c1aa0d3179e1aac23f79fc04", + "url": "https://github.com/pineforge-4pass/pineforge-engine/blob/59082e696f0c7a95c1aa0d3179e1aac23f79fc04/LICENSE" + }, + "codegen": { + "label": "source", + "repo": "pineforge-4pass/pineforge-codegen-oss", + "commit": "1e6eb83354b74ede1326fe55e576e91ea81f156b", + "url": "https://github.com/pineforge-4pass/pineforge-codegen-oss/blob/1e6eb83354b74ede1326fe55e576e91ea81f156b/LICENSE" + }, + "corpus": { + "label": "apache", + "repo": "pineforge-4pass/pineforge-corpus", + "commit": "262bab123176989464cbab04aa94c72b3d3143be", + "url": "https://github.com/pineforge-4pass/pineforge-corpus/blob/262bab123176989464cbab04aa94c72b3d3143be/LICENSE" + } + }, + "wording": { + "cpp_standard": "C++17", + "pine_language": "Pine Script v6", + "pine_compact": "PineScript v6" + }, + "repositories": [ + { + "role": "engine", + "repo": "pineforge-4pass/pineforge-engine", + "public": true, + "disposition": "managed", + "reason": "Approved published stopgap. Graded probes use the current scoreboard. Parenthesized strategy/script counts are historical inventory bound to inventory.sourceRelease, not the graded-probe denominator.", + "source": { + "repo": "pineforge-4pass/pineforge-engine", + "commit": "59082e696f0c7a95c1aa0d3179e1aac23f79fc04", + "url": "https://github.com/pineforge-4pass/pineforge-engine/blob/59082e696f0c7a95c1aa0d3179e1aac23f79fc04/LICENSE" + }, + "template": "Open-source {{word:cpp_standard}} engine for backtesting and forward execution, with a versioned C ABI; {{word:pine_language}} runs through code generation. Against TradingView's own trade lists, all {{facts:scoreboard.graded|grouped}} graded probes ({{facts:inventory.corpusScripts|int}} open-corpus strategies, {{facts:inventory.closedScripts|int}} closed-test scripts) grade excellent or strong: {{facts:scoreboard.excellent|grouped}} excellent, {{facts:scoreboard.strong|int}} strong. {{license:engine}}." + }, + { + "role": "codegen-oss", + "repo": "pineforge-4pass/pineforge-codegen-oss", + "public": true, + "disposition": "managed", + "reason": "Approved published stopgap; retain the verified current license label and existing product wording without a new licensing decision.", + "source": { + "repo": "pineforge-4pass/pineforge-codegen-oss", + "commit": "1e6eb83354b74ede1326fe55e576e91ea81f156b", + "url": "https://github.com/pineforge-4pass/pineforge-codegen-oss/blob/1e6eb83354b74ede1326fe55e576e91ea81f156b/LICENSE" + }, + "template": "{{word:pine_language}} to C++ transpiler for the pineforge-engine runtime. Pure Python; source-available under the {{license:codegen}}: free for noncommercial use and personal trading, commercial use needs a license." + }, + { + "role": "corpus", + "repo": "pineforge-4pass/pineforge-corpus", + "public": true, + "disposition": "managed", + "reason": "Replace stale untracked quantities. Explicitly distinguish current graded corpus probes from historical authored-script inventory; omit unsupported trade totals.", + "source": { + "repo": "pineforge-4pass/pineforge-corpus", + "commit": "262bab123176989464cbab04aa94c72b3d3143be", + "url": "https://github.com/pineforge-4pass/pineforge-corpus/blob/262bab123176989464cbab04aa94c72b3d3143be/LICENSE" + }, + "template": "Open Pine Script reference corpus and TradingView trade traces. Current graded corpus: {{facts:scoreboard.scopes.corpus.excellent|grouped}}/{{facts:scoreboard.scopes.corpus.graded|grouped}} excellent probes. Historical script inventory ({{facts:inventory.sourceRelease|text}}): {{facts:inventory.corpusScripts|grouped}} scripts. {{license:corpus}}." + }, + { + "role": "hpo", + "repo": "pineforge-4pass/pineforge-hpo", + "public": true, + "disposition": "HOLD", + "reason": "Published description is held/read-only pending authorization. No new license claim, release/catalog change, or setting-change command.", + "source": { + "repo": "pineforge-4pass/pineforge-hpo", + "commit": "d66788a6160b271d78270652acf7f8f82d906343", + "url": "https://github.com/pineforge-4pass/pineforge-hpo/blob/d66788a6160b271d78270652acf7f8f82d906343/LICENSE" + }, + "approved_text": "Native C++ hyperparameter optimization for PineForge strategies: grid, seeded random, dlib global search, and TPE (alpha)" + }, + { + "role": "release", + "repo": "pineforge-4pass/pineforge-release", + "public": true, + "disposition": "static", + "reason": "Preserve verified published product wording; no quantitative parity claim.", + "source": { + "repo": "pineforge-4pass/pineforge-release", + "commit": "f93dfca830b8a4a20d43c682a84d52a62bd942fd", + "url": "https://github.com/pineforge-4pass/pineforge-release/blob/f93dfca830b8a4a20d43c682a84d52a62bd942fd/LICENSE" + }, + "template": "Full {{word:pine_compact}} → deterministic backtest: pineforge-engine runtime + bundled codegen transpiler" + }, + { + "role": "backtest-mcp", + "repo": "pineforge-4pass/pineforge-backtest-mcp", + "public": true, + "disposition": "static", + "reason": "Preserve verified published product wording; no quantitative parity claim.", + "source": { + "repo": "pineforge-4pass/pineforge-backtest-mcp", + "commit": "7ced51880a47af6793e80c65d5de0599c251e25a", + "url": "https://github.com/pineforge-4pass/pineforge-backtest-mcp/blob/7ced51880a47af6793e80c65d5de0599c251e25a/LICENSE" + }, + "template": "Offline MCP server: an AI agent transpiles {{word:pine_compact}} → C++ and runs deterministic, TradingView-parity backtests locally (transpile, backtest, parameter grid, Binance OHLCV, Pine coverage lookup). One Docker container, no API key; code & data stay on your machine." + } + ] +} diff --git a/scripts/repo_descriptions.py b/scripts/repo_descriptions.py new file mode 100644 index 0000000..2d77747 --- /dev/null +++ b/scripts/repo_descriptions.py @@ -0,0 +1,351 @@ +#!/usr/bin/env python3 +"""Render public repository descriptions and check them with read-only GETs.""" +from __future__ import annotations + +import argparse +import datetime +import hashlib +import json +import math +import os +import re +import shlex +import shutil +import subprocess +import sys +import unicodedata +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +DEFAULT_POLICY = ROOT / "facts" / "repo-descriptions.json" +FACTS_SCHEMA = ROOT / "facts" / "facts.schema.json" +API_TIMEOUT_SECONDS = 60 +MAX_INPUT_BYTES = 2 * 1024 * 1024 +ROLES = {"engine", "codegen-oss", "corpus", "hpo", "release", "backtest-mcp"} +REPOSITORY = re.compile(r"[A-Za-z0-9][A-Za-z0-9-]*/[A-Za-z0-9][A-Za-z0-9_.-]*\Z") +COMMIT = re.compile(r"[a-f0-9]{40}\Z") +TOKEN = re.compile(r"\{\{(facts|license|word):([A-Za-z0-9_.-]+)(?:\|(int|grouped|decimal|text))?\}\}") + + +class InputError(ValueError): + pass + + +def unique_object(pairs): + result = {} + for key, value in pairs: + if key in result: + raise InputError(f"duplicate JSON key: {key}") + result[key] = value + return result + + +def decode(raw, name): + def invalid_constant(value): + raise InputError(f"non-finite JSON number: {value}") + try: + return json.loads(raw.decode("utf-8"), object_pairs_hook=unique_object, + parse_constant=invalid_constant) + except (UnicodeError, ValueError, RecursionError) as error: + raise InputError(f"{name}: {error}") from error + + +def read_json(path): + with Path(path).open("rb") as stream: + raw = stream.read(MAX_INPUT_BYTES + 1) + if len(raw) > MAX_INPUT_BYTES: + raise InputError(f"input exceeds {MAX_INPUT_BYTES} bytes: {path}") + return decode(raw, str(path)), hashlib.sha256(raw).hexdigest() + + +def validate_schema(value, schema, root, location="facts"): + supported = {"$schema", "$id", "title", "$defs", "$ref", "type", "const", "anyOf", + "required", "properties", "additionalProperties", "propertyNames", "items", + "pattern", "minimum", "maximum", "format"} + if set(schema) - supported: + raise InputError("unsupported facts schema keyword; upgrade the validator explicitly") + if "$ref" in schema: + reference = schema["$ref"] + if not reference.startswith("#/$defs/"): + raise InputError("unsupported schema reference") + validate_schema(value, root["$defs"][reference[8:]], root, location) + if "anyOf" in schema: + for choice in schema["anyOf"]: + try: + validate_schema(value, choice, root, location) + break + except InputError: + continue + else: + raise InputError(f"{location}: does not match any allowed type") + types = {"object": isinstance(value, dict), "array": isinstance(value, list), + "string": isinstance(value, str), "integer": type(value) is int, + "number": type(value) in (int, float), "null": value is None} + if "type" in schema and not types.get(schema["type"], False): + raise InputError(f"{location}: expected {schema['type']}") + if "const" in schema and value != schema["const"]: + raise InputError(f"{location}: incorrect schema version") + if type(value) in (int, float): + if isinstance(value, float) and not math.isfinite(value): + raise InputError(f"{location}: non-finite number") + if "minimum" in schema and value < schema["minimum"]: + raise InputError(f"{location}: below minimum") + if "maximum" in schema and value > schema["maximum"]: + raise InputError(f"{location}: above maximum") + if isinstance(value, str): + if "pattern" in schema and not re.search(schema["pattern"], value, re.ASCII): + raise InputError(f"{location}: invalid spelling") + if "format" in schema: + try: + if schema["format"] == "date": + if not re.fullmatch(r"\d{4}-\d{2}-\d{2}", value, re.ASCII): + raise ValueError("invalid date spelling") + datetime.date.fromisoformat(value) + elif schema["format"] == "date-time": + if not re.fullmatch(r"\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})", value, re.ASCII): + raise ValueError("invalid timestamp spelling") + datetime.datetime.fromisoformat(value.replace("Z", "+00:00")) + else: + raise ValueError("unsupported format") + except ValueError as error: + raise InputError(f"{location}: {error}") from error + if isinstance(value, dict): + if set(schema.get("required", [])) - set(value): + raise InputError(f"{location}: missing required fields") + properties = schema.get("properties", {}) + for key, child in value.items(): + if "propertyNames" in schema: + validate_schema(key, schema["propertyNames"], root, location) + rule = properties.get(key, schema.get("additionalProperties", True)) + if rule is False: + raise InputError(f"{location}: unknown field {key}") + if isinstance(rule, dict): + validate_schema(child, rule, root, location + "." + key) + if isinstance(value, list) and "items" in schema: + for index, child in enumerate(value): + validate_schema(child, schema["items"], root, f"{location}[{index}]") + + +def fields(value, required, optional=()): + if not isinstance(value, dict) or set(required) - set(value) or set(value) - set(required) - set(optional): + raise InputError("missing or unknown policy fields") + + +def clean_text(value, *, description=False): + if not isinstance(value, str) or not value.strip() or any( + unicodedata.category(character).startswith("C") or character in "\u2028\u2029" + for character in value + ): + raise InputError("empty text or newline/control character") + if description and len(value) > 350: + raise InputError("GitHub descriptions must not exceed 350 characters") + return value + + +def repository_identity(value): + if not isinstance(value, str) or not REPOSITORY.fullmatch(value): + raise InputError("invalid public repository identity") + return value.casefold() + + +def source(value): + fields(value, {"repo", "commit", "url"}) + identity = repository_identity(value["repo"]) + if not isinstance(value["commit"], str) or not COMMIT.fullmatch(value["commit"]): + raise InputError("source must pin a full public commit") + prefix, suffix = "https://github.com/", f"/blob/{value['commit']}/LICENSE" + url = value["url"] + if (not isinstance(url, str) or not url.startswith(prefix) or not url.endswith(suffix) + or repository_identity(url[len(prefix):-len(suffix)]) != identity): + raise InputError("LICENSE source URL does not match its repository/commit") + + +def flatten(value, prefix="", result=None): + if result is None: + result = {} + if isinstance(value, dict): + for key, child in value.items(): + flatten(child, prefix + "." + key if prefix else key, result) + elif not isinstance(value, list): + if prefix in result: + raise InputError("ambiguous facts token path") + result[prefix] = value + return result + + +def render_template(template, facts, policy): + clean_text(template) + names, licenses = set(), set() + + def replace(match): + kind, name, formatting = match.groups() + if kind == "facts": + if name not in facts: + raise InputError(f"unknown or missing facts token: {name}") + value = facts[name] + names.add(name) + if formatting in {"int", "grouped"} and type(value) is int and value >= 0: + return format(value, ",d") if formatting == "grouped" else str(value) + if formatting == "decimal" and type(value) in (int, float) and value >= 0 and math.isfinite(value): + return str(value) + if formatting == "text" and isinstance(value, str): + return clean_text(value) + raise InputError(f"invalid token type or format: {name}") + if formatting is not None: + raise InputError("only facts tokens take a format") + if kind == "license": + if name not in policy["licenses"]: + raise InputError(f"unknown license token: {name}") + licenses.add(name) + return policy["labels"][policy["licenses"][name]["label"]] + if name not in policy["wording"]: + raise InputError(f"unknown wording token: {name}") + return policy["wording"][name] + + literals = TOKEN.sub("", template) + if "{" in literals or "}" in literals or any(character.isdecimal() for character in literals): + raise InputError("unresolved token or literal numeric claim in template") + expected = clean_text(TOKEN.sub(replace, template), description=True) + if "{{" in expected or "}}" in expected: + raise InputError("unresolved token in description") + sources = [dict(id=name, label=policy["labels"][policy["licenses"][name]["label"]], + **{key: policy["licenses"][name][key] for key in ("repo", "commit", "url")}) + for name in sorted(licenses)] + return expected, sorted(names), sources + + +def render(facts_path, policy_path): + facts, facts_hash = read_json(facts_path) + schema, _ = read_json(FACTS_SCHEMA) + validate_schema(facts, schema, schema) + scoreboard = facts["scoreboard"] + if (scoreboard["belowStrong"] != 0 + or scoreboard["excellent"] + scoreboard["strong"] != scoreboard["graded"]): + raise InputError("engine all-graded claim requires belowStrong == 0 and excellent + strong == graded") + policy, policy_hash = read_json(policy_path) + fields(policy, {"schema", "version", "labels", "licenses", "wording", "repositories"}) + if policy["schema"] != "pineforge/repo-descriptions/v1" or type(policy["version"]) is not int or policy["version"] < 1: + raise InputError("invalid description policy version") + for mapping in (policy["labels"], policy["licenses"], policy["wording"]): + if not isinstance(mapping, dict): + raise InputError("policy maps must be objects") + for value in [*policy["labels"].values(), *policy["wording"].values()]: + clean_text(value) + for license_value in policy["licenses"].values(): + fields(license_value, {"label", "repo", "commit", "url"}) + source({key: license_value[key] for key in ("repo", "commit", "url")}) + if license_value["label"] not in policy["labels"]: + raise InputError("unknown approved license label") + if not isinstance(policy["repositories"], list): + raise InputError("repository manifest must be an array") + seen_roles, seen_repos, rows = set(), set(), [] + for row in policy["repositories"]: + fields(row, {"role", "repo", "public", "disposition", "reason", "source"}, {"template", "approved_text"}) + source(row["source"]) + identity = repository_identity(row["repo"]) + if identity != repository_identity(row["source"]["repo"]) or row["public"] is not True: + raise InputError("unverified or nonpublic repository") + if row["role"] not in ROLES or row["role"] in seen_roles or identity in seen_repos: + raise InputError("unknown or duplicate repository/role") + seen_roles.add(row["role"]) + seen_repos.add(identity) + clean_text(row["reason"]) + disposition = row["disposition"] + if (row["role"] == "hpo" or identity.split("/")[1] == "pineforge-hpo") and disposition != "HOLD": + raise InputError("HPO must remain held; no setting-change command is permitted") + if disposition == "HOLD": + if "template" in row or "approved_text" not in row: + raise InputError("held rows require only approved_text") + expected = row["approved_text"] + if expected is not None: + clean_text(expected, description=True) + tokens, sources = [], [] + elif disposition in {"managed", "static"}: + if "approved_text" in row or "template" not in row: + raise InputError("managed/static rows require only a template") + expected, tokens, sources = render_template(row["template"], flatten(facts), policy) + if disposition == "static" and tokens: + raise InputError("static rows cannot contain facts claims") + else: + raise InputError("unknown disposition") + rows.append(dict(role=row["role"], repo=row["repo"], disposition=disposition, + reason=row["reason"], expected=expected, tokens=tokens, + license_sources=sources, public_source=row["source"])) + if seen_roles != ROLES: + raise InputError("incomplete confirmed-public manifest") + return dict(schema="pineforge/repo-descriptions-render/v1", policy_version=policy["version"], + facts_sha256=facts_hash, policy_sha256=policy_hash, repositories=rows) + + +def check_live(document, executable): + resolved = shutil.which(executable) + if not resolved or not os.access(resolved, os.X_OK): + raise InputError("GitHub executable is missing or not executable") + exit_code = 0 + for row in document["repositories"]: + try: + process = subprocess.run( + [resolved, "api", "--method", "GET", "--hostname", "github.com", "repos/" + row["repo"]], + capture_output=True, timeout=API_TIMEOUT_SECONDS, check=False, + ) + if process.returncode: + raise InputError(f"GitHub API/auth error (exit {process.returncode}): " + + process.stderr.decode("utf-8", errors="replace")[:500]) + if len(process.stdout) > MAX_INPUT_BYTES: + raise InputError("GitHub response exceeds size limit") + response = decode(process.stdout, "GitHub response") + if (not isinstance(response, dict) or response.get("private") is not False + or repository_identity(response.get("full_name")) != repository_identity(row["repo"])): + raise InputError("GitHub did not confirm the exact public repository") + if "description" not in response or not (response["description"] is None or isinstance(response["description"], str)): + raise InputError("GitHub description is missing or has the wrong type") + row["actual"] = response["description"] + row["status"] = "match" if row["actual"] == row["expected"] else "drift" + if row["status"] == "drift": + exit_code = max(exit_code, 1) + except (InputError, OSError, subprocess.TimeoutExpired) as error: + row["status"], row["error"] = "error", str(error) + exit_code = 2 + document["exit_code"] = exit_code + return exit_code + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__, allow_abbrev=False) + commands = parser.add_subparsers(dest="command", required=True) + for name in ("render", "check"): + command = commands.add_parser(name, allow_abbrev=False) + command.add_argument("--facts", required=True, type=Path) + command.add_argument("--policy", default=DEFAULT_POLICY, type=Path) + if name == "render": + command.add_argument("--format", choices=("json", "commands"), default="json") + else: + command.add_argument("--live", required=True, action="store_true") + command.add_argument("--gh", default="gh") + raw_arguments = sys.argv[1:] if argv is None else argv + options = [value.split("=", 1)[0] for value in raw_arguments if value.startswith("--")] + if len(options) != len(set(options)): + parser.error("repeated options are ambiguous") + arguments = parser.parse_args(raw_arguments) + try: + document = render(arguments.facts, arguments.policy) + exit_code = check_live(document, arguments.gh) if arguments.command == "check" else 0 + if arguments.command == "render" and arguments.format == "commands": + lines = ["# TOP review only; this program never applies descriptions.", + "# facts_sha256=" + document["facts_sha256"], + "# policy_sha256=" + document["policy_sha256"]] + for row in document["repositories"]: + lines.append("# " + row["disposition"] + " " + json.dumps(row, ensure_ascii=True, sort_keys=True)) + if row["disposition"] != "HOLD": + lines.append(shlex.join(["gh", "repo", "edit", row["repo"], "--description", row["expected"]])) + print("\n".join(lines)) + else: + print(json.dumps(document, indent=2, ensure_ascii=True, sort_keys=True)) + return exit_code + except (InputError, OSError, ValueError, KeyError, TypeError, RecursionError, OverflowError) as error: + print(f"repo-descriptions: {error}", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_repo_descriptions.py b/tests/test_repo_descriptions.py new file mode 100644 index 0000000..9c81c7f --- /dev/null +++ b/tests/test_repo_descriptions.py @@ -0,0 +1,379 @@ +#!/usr/bin/env python3 +"""Offline CLI acceptance; only the third-party GitHub executable is faked.""" +from __future__ import annotations + +import copy +import hashlib +import json +import os +import shlex +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +REPO = Path(__file__).resolve().parent.parent +SCRIPT = REPO / "scripts" / "repo_descriptions.py" +FACTS = REPO / "facts" / "facts.json" +POLICY = REPO / "facts" / "repo-descriptions.json" + + +class DescriptionCLI(unittest.TestCase): + def setUp(self): + self.temporary = tempfile.TemporaryDirectory(prefix="repo-description-acceptance-") + self.addCleanup(self.temporary.cleanup) + self.root = Path(self.temporary.name) + + def invoke(self, *arguments, facts=FACTS, policy=POLICY, env=None, timeout=100): + return subprocess.run( + [sys.executable, str(SCRIPT), *arguments, "--facts", str(facts), + "--policy", str(policy)], + capture_output=True, text=True, check=False, timeout=timeout, env=env, + ) + + def document(self, name, value): + path = self.root / name + path.write_text(json.dumps(value), encoding="utf-8") + return path + + def policy(self, change): + policy = json.loads(POLICY.read_text(encoding="utf-8")) + change(policy) + return self.document("policy.json", policy) + + def render(self, **kwargs): + result = self.invoke("render", **kwargs) + self.assertEqual(result.returncode, 0, result.stderr) + return json.loads(result.stdout) + + def test_real_canonical_snapshot_and_complete_manifest(self): + document = self.render() + self.assertEqual(document["facts_sha256"], hashlib.sha256(FACTS.read_bytes()).hexdigest()) + rows = {row["role"]: row for row in document["repositories"]} + self.assertEqual(set(rows), {"engine", "codegen-oss", "corpus", "hpo", "release", "backtest-mcp"}) + self.assertEqual(rows["hpo"]["disposition"], "HOLD") + self.assertIn("all 7,989 graded probes", rows["engine"]["expected"]) + self.assertIn("7,983 excellent, 6 strong", rows["engine"]["expected"]) + self.assertIn("6 strong", rows["engine"]["expected"]) + self.assertIn("309 open-corpus strategies, 741 closed-test scripts", + rows["engine"]["expected"]) + self.assertIn("Historical script inventory (1.0.1): 309 scripts", rows["corpus"]["expected"]) + self.assertIn("PineForge Source License 1.2", rows["codegen-oss"]["expected"]) + self.assertIn("scoreboard.graded", rows["engine"]["tokens"]) + self.assertIn("inventory.closedScripts", rows["engine"]["tokens"]) + for row in rows.values(): + for source in row["license_sources"]: + self.assertIn("/" + source["commit"] + "/LICENSE", source["url"]) + self.assertEqual(self.invoke("render").stdout, self.invoke("render").stdout) + + def test_future_main_does_not_relabel_inventory_or_released_metrics(self): + facts = json.loads(FACTS.read_text(encoding="utf-8")) + facts["scoreboard"].update(graded=9000, excellent=8998, strong=2, population=9017) + future = self.document("future.json", facts) + rows = {row["role"]: row for row in self.render(facts=future)["repositories"]} + self.assertIn("all 9,000 graded probes", rows["engine"]["expected"]) + self.assertIn("8,998 excellent, 2 strong", rows["engine"]["expected"]) + self.assertIn("309 open-corpus strategies, 741 closed-test scripts", rows["engine"]["expected"]) + policy = self.policy(lambda value: value["repositories"][0].update( + template="Release probes: {{facts:releases.1.0.1.scoreboard.graded|int}}; " + "current: {{facts:scoreboard.graded|int}}; " + "date: {{facts:scoreboard.date|text}}; " + "engine: {{facts:scoreboard.engineCommit|text}}.")) + engine = self.render(facts=future, policy=policy)["repositories"][0] + self.assertEqual(engine["expected"], + "Release probes: 7989; current: 9000; date: 2026-10-06; " + "engine: 59082e696f0c7a95c1aa0d3179e1aac23f79fc04.") + self.assertNotEqual(self.render(facts=future)["facts_sha256"], + self.render()["facts_sha256"]) + + def test_consistent_below_strong_snapshot_is_refused_before_output_or_live_reads(self): + canonical = json.loads(FACTS.read_text(encoding="utf-8")) + facts = copy.deepcopy(canonical) + scoreboard = facts["scoreboard"] + pair = next(value for value in scoreboard["pairs"] + if value["hardProbes"] == 0 and value["corpusProbes"] == 0 + and value["closedProbes"] > 0) + for group in (scoreboard, scoreboard["scopes"]["closed"], pair): + group["excellent"] -= 1 + group["belowStrong"] += 1 + group["tiers"]["excellent"] -= 1 + group["tiers"]["moderate"] += 1 + for field, count in (("excellentPct", scoreboard["excellent"]), + ("strongPct", scoreboard["strong"]), + ("excellentOrStrongPct", scoreboard["excellent"] + scoreboard["strong"])): + scoreboard[field] = round(count * 100 / scoreboard["graded"], 2) + self.assertEqual(facts["inventory"], canonical["inventory"]) + self.assertEqual(facts["releases"], canonical["releases"]) + self.assertEqual(scoreboard["hardLane"], canonical["scoreboard"]["hardLane"]) + self.assertEqual(scoreboard["population"] - scoreboard["anomaliesExcluded"], + scoreboard["graded"]) + for group in [scoreboard, *scoreboard["scopes"].values(), *scoreboard["pairs"]]: + self.assertEqual(sum(group["tiers"].values()), group["graded"]) + self.assertEqual(group["excellent"] + group["strong"] + group["belowStrong"], + group["graded"]) + self.assertEqual(group["belowStrong"], sum(value for tier, value in group["tiers"].items() + if tier not in {"excellent", "strong"})) + for field in ("graded", "excellent", "strong", "belowStrong", "engineErrors"): + self.assertEqual(sum(value[field] for value in scoreboard["pairs"]), scoreboard[field]) + self.assertEqual(sum(value[field] for value in scoreboard["scopes"].values()), + scoreboard[field]) + future = self.document("future-below-strong.json", facts) + executable, env, _ = self.fake_gh() + for arguments in [("render",), ("render", "--format", "commands"), + ("check", "--live", "--gh", str(executable))]: + with self.subTest(arguments=arguments): + result = self.invoke(*arguments, facts=future, env=env) + self.assertEqual(result.returncode, 2, result.stdout) + self.assertEqual(result.stdout, "") + self.assertIn("all-graded", result.stderr) + self.assertFalse((self.root / "api-calls.jsonl").exists()) + + def test_all_graded_claim_requires_exact_excellent_and_strong_sum(self): + for difference in (-1, 1): + with self.subTest(difference=difference): + facts = json.loads(FACTS.read_text(encoding="utf-8")) + facts["scoreboard"]["excellent"] += difference + result = self.invoke("render", "--format", "commands", + facts=self.document("inconsistent-sum.json", facts)) + self.assertEqual(result.returncode, 2, result.stdout) + self.assertEqual(result.stdout, "") + self.assertIn("all-graded", result.stderr) + + def test_changed_historical_tokens_are_read_not_hardcoded(self): + facts = json.loads(FACTS.read_text(encoding="utf-8")) + facts["inventory"].update(corpusScripts=411, closedScripts=855) + engine = self.render(facts=self.document("inventory.json", facts))["repositories"][0] + self.assertIn("411 open-corpus strategies, 855 closed-test scripts", engine["expected"]) + + def test_missing_unknown_and_unresolved_tokens_fail_without_partial_commands(self): + for template in ["{{facts:scoreboard.unknown|int}}", "{{facts:inventory|text}}", + "{{license:unknown}}", "{{facts:scoreboard.graded}}", "{{bad}}", + "{{facts:scoreboard.graded|float}}", "{{facts:scoreboard.graded|int}", + "Count: 7989", "{{facts:scoreboard.graded|text}}"]: + with self.subTest(template=template): + policy = self.policy(lambda value: value["repositories"][0].update(template=template)) + result = self.invoke("render", "--format", "commands", policy=policy) + self.assertEqual(result.returncode, 2, result.stdout) + self.assertEqual(result.stdout, "") + facts = json.loads(FACTS.read_text(encoding="utf-8")) + del facts["scoreboard"]["graded"] + result = self.invoke("render", facts=self.document("missing.json", facts)) + self.assertEqual(result.returncode, 2) + + def test_invalid_quantities_and_schema_are_refused(self): + for invalid in [None, True, -1, 1.5, "7989", [], {}]: + with self.subTest(invalid=invalid): + facts = json.loads(FACTS.read_text(encoding="utf-8")) + facts["scoreboard"]["graded"] = invalid + result = self.invoke("render", facts=self.document("invalid.json", facts)) + self.assertEqual(result.returncode, 2, result.stdout) + for change in [lambda value: value.update(schema="other"), + lambda value: value.update(unrecognized=1), + lambda value: value["scoreboard"].update(excellentPct=101), + lambda value: value["scoreboard"].update(date="2026-02-30"), + lambda value: value["scoreboard"]["provenance"].update(promotionDate="yesterday")]: + facts = json.loads(FACTS.read_text(encoding="utf-8")) + change(facts) + self.assertEqual(self.invoke("render", facts=self.document("bad-schema.json", facts)).returncode, 2) + + def test_ambiguous_json_and_nonfinite_numbers_are_refused(self): + for raw in ['{"schema":1,"schema":2}', '{"value":NaN}', '{"value":Infinity}', + '{"value":-Infinity}', '[]', '{} trailing', '\ufeff{}', '{"value":1e9999}']: + with self.subTest(raw=raw): + path = self.root / "bad.json" + path.write_text(raw, encoding="utf-8") + self.assertEqual(self.invoke("render", facts=path).returncode, 2) + path.write_bytes(b"\xff") + self.assertEqual(self.invoke("render", facts=path).returncode, 2) + + def test_github_character_boundary_and_control_characters(self): + for size, expected_exit in [(350, 0), (351, 2)]: + policy = self.policy(lambda value: value["repositories"][0].update(template="é" * size)) + self.assertEqual(self.invoke("render", policy=policy).returncode, expected_exit) + for control in ["\n", "\r", "\t", "\0", "\x1f", "\x7f", "\x85", "\u2028", "\u202e"]: + with self.subTest(control=repr(control)): + policy = self.policy(lambda value: value["repositories"][0].update(template="bad" + control)) + self.assertEqual(self.invoke("render", policy=policy).returncode, 2) + policy = self.policy(lambda value: value["repositories"][0].update(template="")) + self.assertEqual(self.invoke("render", policy=policy).returncode, 2) + policy = self.policy(lambda value: value["repositories"][0].update(template=" ")) + self.assertEqual(self.invoke("render", policy=policy).returncode, 2) + + def test_repeated_cli_options_are_refused(self): + self.assertEqual(self.invoke("render", "--facts", str(FACTS)).returncode, 2) + self.assertEqual(self.invoke("render", "--format=json", "--format", "commands").returncode, 2) + + def test_hpo_cannot_be_unheld_by_swapping_roles(self): + for identity in ("pineforge-4pass/pineforge-hpo", "pineforge-4pass/PineForge-HPO", + "PINEFORGE-4PASS/PINEFORGE-HPO"): + with self.subTest(identity=identity): + def change(policy): + engine = next(row for row in policy["repositories"] if row["role"] == "engine") + held = next(row for row in policy["repositories"] if row["role"] == "hpo") + engine.update(role="hpo", disposition="HOLD", approved_text="Held engine") + del engine["template"] + held["source"]["url"] = held["source"]["url"].replace(held["repo"], identity) + held["source"]["repo"] = identity + held.update(role="engine", repo=identity, disposition="static", + template="Changed held repository") + del held["approved_text"] + result = self.invoke("render", "--format", "commands", policy=self.policy(change)) + self.assertEqual(result.returncode, 2) + self.assertEqual(result.stdout, "") + + def test_case_only_repository_duplicates_fail_without_partial_output(self): + def change(policy): + original, duplicate = policy["repositories"][:2] + identity = original["repo"].swapcase() + duplicate["repo"] = identity + duplicate["source"] = copy.deepcopy(original["source"]) + duplicate["source"]["repo"] = identity + duplicate["source"]["url"] = duplicate["source"]["url"].replace(original["repo"], identity) + result = self.invoke("render", "--format", "commands", policy=self.policy(change)) + self.assertEqual(result.returncode, 2, result.stdout) + self.assertEqual(result.stdout, "") + self.assertIn("duplicate", result.stderr) + + def test_mixed_case_source_and_live_identities_are_accepted(self): + def change(policy): + for row in policy["repositories"]: + row["repo"] = row["repo"].swapcase() + for license_value in policy["licenses"].values(): + license_value["repo"] = license_value["repo"].swapcase() + policy = self.policy(change) + canonical_rows = self.render()["repositories"] + rendered_rows = self.render(policy=policy)["repositories"] + self.assertEqual([row["expected"] for row in canonical_rows], + [row["expected"] for row in rendered_rows]) + executable, env, _ = self.fake_gh("case") + result = self.invoke("check", "--live", "--gh", str(executable), policy=policy, env=env) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertTrue(all(row["status"] == "match" for row in json.loads(result.stdout)["repositories"])) + malformed = self.policy(lambda value: value["licenses"]["engine"].update( + url=value["licenses"]["engine"]["url"].replace("/LICENSE", "/license"))) + self.assertEqual(self.invoke("render", policy=malformed).returncode, 2) + + def test_policy_refuses_duplicate_repositories_nonpublic_and_unpinned_licenses(self): + changes = [lambda value: value["repositories"].append(copy.deepcopy(value["repositories"][0])), + lambda value: value["repositories"].pop(), + lambda value: value["repositories"][0].update(repo="--evil/name"), + lambda value: value["repositories"][0].update(public=False), + lambda value: value["repositories"][0].update(unknown=True), + lambda value: value["licenses"]["codegen"].update(commit="main"), + lambda value: value["licenses"]["codegen"].update(url="https://example.com/LICENSE"), + lambda value: next(row for row in value["repositories"] if row["role"] == "hpo").update( + disposition="managed", template="Mutate HPO")] + for change in changes: + with self.subTest(change=change): + self.assertEqual(self.invoke("render", policy=self.policy(change)).returncode, 2) + + def test_commands_are_quoted_text_and_hpo_is_never_a_command(self): + dangerous = "Compiler's $(touch SHOULD_NOT_EXIST); `id` & shell text" + policy = self.policy(lambda value: value["repositories"][0].update(template=dangerous)) + result = self.invoke("render", "--format", "commands", policy=policy) + self.assertEqual(result.returncode, 0, result.stderr) + commands = [shlex.split(line) for line in result.stdout.splitlines() if not line.startswith("#")] + self.assertEqual(len(commands), 5) + self.assertEqual(commands[0], ["gh", "repo", "edit", "pineforge-4pass/pineforge-engine", + "--description", dangerous]) + self.assertTrue(all(command[3] != "pineforge-4pass/pineforge-hpo" for command in commands)) + self.assertIn("# HOLD", result.stdout) + self.assertIn("facts_sha256", result.stdout) + self.assertFalse((self.root / "SHOULD_NOT_EXIST").exists()) + + def fake_gh(self, mode="match", role="engine"): + rendered = self.render() + rows = rendered["repositories"] + state = {row["repo"]: {"full_name": row["repo"], "private": False, + "description": row["expected"]} for row in rows} + target = next(row["repo"] for row in rows if row["role"] == role) + if mode == "drift": state[target]["description"] += " changed" + if mode == "null": state[target]["description"] = None + if mode == "empty": state[target]["description"] = "" + if mode == "nonpublic": state[target]["private"] = True + if mode == "identity": state[target]["full_name"] = "other/name" + if mode == "type": state[target]["description"] = 42 + if mode == "missing": del state[target]["description"] + if mode == "case": + for response in state.values(): response["full_name"] = response["full_name"].swapcase() + state_path = self.document("live-state.json", state) + executable = self.root / "fake third-party gh" + executable.write_text( + "#!/usr/bin/env python3\n" + "import json, os, sys, time\n" + "from pathlib import Path\n" + "args = sys.argv[1:]\n" + "with open(os.environ['FAKE_GH_LOG'], 'a') as output: output.write(json.dumps(args) + '\\n')\n" + "if args[:5] != ['api', '--method', 'GET', '--hostname', 'github.com'] or len(args) != 6:\n" + " print('refusing non-read operation', file=sys.stderr); sys.exit(80)\n" + "repo = args[5].removeprefix('repos/').casefold()\n" + "if repo == os.environ['FAKE_GH_TARGET']:\n" + " mode = os.environ['FAKE_GH_MODE']\n" + " if mode == 'error': print('authentication failed', file=sys.stderr); sys.exit(7)\n" + " if mode == 'malformed': print('not JSON'); sys.exit(0)\n" + " if mode == 'timeout': time.sleep(65)\n" + "state = json.loads(Path(os.environ['FAKE_GH_STATE']).read_text())\n" + "print(json.dumps(state[repo]))\n", encoding="utf-8") + executable.chmod(0o755) + env = dict(os.environ, FAKE_GH_STATE=str(state_path), FAKE_GH_LOG=str(self.root / "api-calls.jsonl"), + FAKE_GH_TARGET=target, FAKE_GH_MODE=mode) + return executable, env, target + + def test_exact_live_match_and_bounded_read_only_arguments(self): + executable, env, _ = self.fake_gh() + result = self.invoke("check", "--live", "--gh", str(executable), env=env) + self.assertEqual(result.returncode, 0, result.stderr) + rows = json.loads(result.stdout)["repositories"] + self.assertEqual(len(rows), 6) + self.assertTrue(all(row["status"] == "match" for row in rows)) + calls = [json.loads(line) for line in (self.root / "api-calls.jsonl").read_text().splitlines()] + self.assertEqual(len(calls), 6) + self.assertTrue(all(call[:5] == ["api", "--method", "GET", "--hostname", "github.com"] for call in calls)) + + def test_live_drift_including_held_hpo_and_explicit_null(self): + for mode, role in [("drift", "engine"), ("drift", "hpo"), ("null", "engine"), ("empty", "engine")]: + with self.subTest(mode=mode, role=role): + executable, env, target = self.fake_gh(mode, role) + result = self.invoke("check", "--live", "--gh", str(executable), env=env) + self.assertEqual(result.returncode, 1, result.stderr) + row = next(row for row in json.loads(result.stdout)["repositories"] if row["repo"] == target) + self.assertEqual(row["status"], "drift") + if mode == "null": self.assertIsNone(row["actual"]) + if mode == "empty": self.assertEqual(row["actual"], "") + self.assertNotEqual(row["actual"], row["expected"]) + + def test_live_api_auth_shape_identity_and_timeout_fail_closed(self): + for mode in ["error", "malformed", "nonpublic", "identity", "type", "missing", "timeout"]: + with self.subTest(mode=mode): + executable, env, target = self.fake_gh(mode) + result = self.invoke("check", "--live", "--gh", str(executable), env=env) + self.assertEqual(result.returncode, 2, result.stdout) + row = next(row for row in json.loads(result.stdout)["repositories"] if row["repo"] == target) + self.assertEqual(row["status"], "error") + self.assertNotIn("actual", row) + self.assertTrue(row["error"]) + self.assertEqual(self.invoke("check", "--live", "--gh", str(self.root / "absent")).returncode, 2) + self.assertEqual(self.invoke("check").returncode, 2) + self.assertEqual(self.invoke("apply").returncode, 2) + + +class ReadOnlyWorkflow(unittest.TestCase): + def test_live_workflow_is_main_only_read_only_and_always_preserves_proposals(self): + text = (REPO / ".github" / "workflows" / "repo-description-drift.yml").read_text(encoding="utf-8") + self.assertIn("branches: [main]", text) + self.assertIn("workflow_dispatch:", text) + self.assertNotIn("pull_request:", text) + self.assertIn("contents: read", text) + self.assertNotIn(": write", text) + for path in ["facts/facts.json", "facts/facts.schema.json", "facts/repo-descriptions.json", + "scripts/repo_descriptions.py"]: + self.assertIn(path, text) + self.assertIn("check --facts facts/facts.json --live", text) + self.assertIn("actions/upload-artifact@", text) + self.assertIn("if: always()", text) + self.assertNotIn("gh repo edit", text) + + +if __name__ == "__main__": + unittest.main()