From bce2808e56e5d4807f249eb61ff4184bba5b38f3 Mon Sep 17 00:00:00 2001 From: abrichr Date: Mon, 27 Jul 2026 23:05:11 -0400 Subject: [PATCH] chore(ci): derive the source-boundary guard from the policy manifest The open-core boundary is described by a machine-readable manifest (source-policy.yaml, canonical in the private openadapt-internal repo), but nothing read it. This guard kept its own hand-copied denylist of the same crown-jewel categories, so the manifest advised while a separate copy enforced, and the two could drift silently. The rules now come from source-policy.public.json: a generated file rendered from the canonical manifest, carrying only the publishable subset (category names, public repository classifications, and the match rules). A public CI job cannot read a private manifest and must never be given a way to, so the rules travel outward as a generated artifact instead. The private repository re-renders and compares this copy daily. The guard fails closed. A missing, unparseable, incomplete, or unknown-schema policy exits 2 and stops the build; a guard that passes with no rules is worse than the hardcoded list it replaced. An unclassified repository also exits 2, which makes the manifest's coverage rule mechanical. Nothing got weaker. tests/test_source_boundary.py pins the exact previous denylist as a one-way ratchet and proves each entry still blocks, proves every policy token is actually enforced, and proves each fail-closed path. The guard additionally gained the private path segments and the private-artifact content banner, which it did not check before. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01NyCHrzA1psrKMFfroYbzaM --- .github/workflows/platform-manifest.yml | 11 + CONTRIBUTING.md | 5 + scripts/check_source_boundary.py | 339 +++++++++++++------ source-policy.public.json | 417 ++++++++++++++++++++++++ tests/test_source_boundary.py | 227 +++++++++++++ 5 files changed, 908 insertions(+), 91 deletions(-) create mode 100644 source-policy.public.json create mode 100644 tests/test_source_boundary.py diff --git a/.github/workflows/platform-manifest.yml b/.github/workflows/platform-manifest.yml index 1bccba23c..96ed10dcb 100644 --- a/.github/workflows/platform-manifest.yml +++ b/.github/workflows/platform-manifest.yml @@ -130,5 +130,16 @@ jobs: with: python-version: '3.12' + # The guard reads its rules from source-policy.public.json (rendered from + # the canonical private manifest) and fails closed when that file is + # missing or incomplete. Stdlib only: no dependency install. - name: Check open-core source boundary run: python scripts/check_source_boundary.py + + # Prove the guard still blocks what it blocked before the rules moved out + # of the script, and that it fails closed without them. + - name: Prove the boundary guard fails when it should + if: github.event_name == 'pull_request' || github.event_name == 'push' + run: | + python -m pip install --quiet 'pytest>=8.0.0' + python -m pytest tests/test_source_boundary.py -q diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 84d663202..8c50cfecf 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -66,6 +66,11 @@ What this means for contributions to the public repos: proprietary system identifiers derived from real deployments (for example, automation content tied to a specific customer's EHR configuration). CI runs `scripts/check_source_boundary.py` to reject that class of content. + That script keeps no denylist of its own: it reads + [`source-policy.public.json`](source-policy.public.json), a generated file + rendered from OpenAdapt's canonical source-availability manifest. Do not edit + the generated file by hand; a rule changes at its source, and the guard fails + closed if the file is missing or incomplete. - Do not contribute deployment-derived corpora, tuned adversary parameters, thresholds, oracle or connector recipes, or datasets tied to real systems of record. Synthetic, reproducible fixtures are welcome. diff --git a/scripts/check_source_boundary.py b/scripts/check_source_boundary.py index 39f516357..a859a6e09 100644 --- a/scripts/check_source_boundary.py +++ b/scripts/check_source_boundary.py @@ -7,94 +7,72 @@ reached a public core surface and had to be excised by hand after review caught it. Application-specific recipes, customer fixtures, and proprietary system identifiers derived from real deployments are exactly the class of -content the open-core boundary keeps private (see the workspace -source-policy.yaml: grown corpora, tuned adversary parameters, -deployment-derived thresholds, oracle/connector recipes, and real-EMR-tied -datasets are crown jewels). This check makes that class of leak a CI failure -instead of a code-review coin flip. +content the open-core boundary keeps private (grown corpora, tuned adversary +parameters, deployment-derived thresholds, oracle/connector recipes, and +real-EMR-tied datasets are crown jewels). This check makes that class of leak +a CI failure instead of a code-review coin flip. -It complements, and deliberately overlaps with, the packaging-time guard in -openadapt-flow's scripts/check_release_consistency.py (which inspects built -wheels/sdists). This script inspects the REPOSITORY TREE, so the leak is -caught at PR time in any public core repo, not only when an artifact is -built. +Where the rules come from +------------------------- +This script owns no denylist. Every token, pattern, and signature it enforces +is read from `source-policy.public.json` in this repository, which is RENDERED +from the canonical source-availability manifest +`OpenAdaptAI/openadapt-internal:source-policy.yaml`. That manifest is private +and a public CI job must never be able to read it, so the publishable subset of +it travels outward as a generated file instead. To change a rule, change the +canonical manifest and re-render; a job in the private repository compares this +copy against the canonical one daily and fails while they disagree. + +The script therefore FAILS CLOSED. A missing, unparseable, or incomplete +`source-policy.public.json` exits 2 and stops the build. A guard that passes +because it found no rules is worse than the hardcoded list it replaced, because +everyone believes it ran. Two kinds of matching, so a rename cannot dodge the check: -* Path tokens: any tracked file whose path contains a denylisted token fails. -* Content signatures: any tracked text file whose contents match a - denylisted pattern fails. +* Path tokens and private path segments: any tracked file whose path contains a + denylisted token, or lies under a private segment, fails. +* Content signatures and patterns: any tracked text file whose contents match a + denylisted regular expression, or carry a private-artifact banner, fails. + +It complements, and deliberately overlaps with, the packaging-time guard in +openadapt-flow's scripts/check_release_consistency.py (which inspects built +wheels/sdists and reads the same rendered policy). This script inspects the +REPOSITORY TREE, so the leak is caught at PR time in any public core repo, not +only when an artifact is built. -Files that legitimately DISCUSS the boundary (this script, policy docs, -contributor docs) are covered by the allowlist below. +Files that legitimately DISCUSS the boundary (this script, the rendered policy +itself, policy docs, contributor docs) are covered by the allowlist below. That +allowlist is repository-local knowledge -- which files here are about the +boundary -- and is deliberately not part of the shared policy. Usage: - python scripts/check_source_boundary.py [--root PATH] + python scripts/check_source_boundary.py [--root PATH] [--repo NAME] --root defaults to this repository. Point it at another checkout to run the -same gate in that repo's CI without vendoring a divergent copy. +same gate in that repo's CI without vendoring a divergent copy; the rules are +always read from THIS checkout, never from the tree being scanned. """ from __future__ import annotations import argparse +import json import re import subprocess import sys from pathlib import Path DEFAULT_ROOT = Path(__file__).resolve().parents[1] - -# Path fragments that indicate private-boundary or deployment-specific -# content. Lowercase; matched case-insensitively against the full repo- -# relative path. -DENYLISTED_PATH_TOKENS = ( - # Proprietary systems of record: per-system recipes/fixtures stay private. - "powerchart", - "cerner", - "millennium_recipe", - "epic_hyperspace", - "meditech", - # Customer/deployment-derived material. - "customer_fixture", - "customer_recipe", - "client_fixture", - "deployment_corpus", - "deployment_thresholds", - "real_emr", - # Crown-jewel categories from source-policy.yaml. - "grown_corpus", - "tuned_adversary", - "adversary_corpus", - "identity_roc", - "held_out_corpus", - "oracle_recipe", - "effect_oracle_recipe", - "pixel_verify_cert", - "enterprise_productionized", - "paid_agent_evidence", - # The private repos themselves must never be vendored into a public one. - "openadapt-corpus", - "openadapt-cloud", - "openadapt-internal", -) - -# Content signatures for text files. Case-insensitive regular expressions. -DENYLISTED_CONTENT_PATTERNS = ( - # Proprietary EHR surfaces showing up inside code, fixtures, or recipes. - r"powerchart", - r"cerner\s+millennium", - r"epic\s+hyperspace", - # Markers our private artifacts carry. - r"openadapt[_-]corpus[/:]", - r"deployment[_-]derived\s+threshold\s*=", - r"oracle[_-]recipe[_-]id\s*[:=]", -) +POLICY_PATH = DEFAULT_ROOT / "source-policy.public.json" +POLICY_SCHEMA_VERSION = 1 # Repo-relative paths (exact file or directory prefix) that may mention the # denylisted terms because they document or enforce the boundary itself. ALLOWLISTED_PATHS = ( "scripts/check_source_boundary.py", + "source-policy.public.json", + "tests/test_source_boundary.py", "docs/platform-manifest.md", "CONTRIBUTING.md", "TRADEMARKS.md", @@ -124,6 +102,108 @@ MAX_CONTENT_BYTES = 5 * 1024 * 1024 +class PolicyError(RuntimeError): + """The rendered policy is missing, unparseable, or incomplete.""" + + +def _require_strings( + container: dict, key: str, *, where: str, lower: bool = True +) -> list[str]: + value = container.get(key) + if not isinstance(value, list) or not value: + raise PolicyError(f"{where}.{key} must be a non-empty list") + items = [] + for item in value: + if not isinstance(item, str) or not item.strip(): + raise PolicyError(f"{where}.{key} must contain non-empty strings") + items.append(item.lower() if lower else item) + return items + + +class SourcePolicy: + """The subset of the rendered manifest this guard enforces.""" + + def __init__(self, document: dict) -> None: + if document.get("schema_version") != POLICY_SCHEMA_VERSION: + raise PolicyError( + f"schema_version is {document.get('schema_version')!r}, expected " + f"{POLICY_SCHEMA_VERSION}; this guard refuses to enforce an " + "unknown policy schema" + ) + enforcement = document.get("enforcement") + if not isinstance(enforcement, dict): + raise PolicyError("enforcement: block is missing") + + self.path_tokens = tuple( + _require_strings(enforcement, "path_tokens", where="enforcement") + ) + self.private_path_segments = frozenset( + _require_strings(enforcement, "private_path_segments", where="enforcement") + ) + + tree = enforcement.get("repository_tree") + if not isinstance(tree, dict): + raise PolicyError("enforcement.repository_tree: block is missing") + patterns = _require_strings( + tree, "content_patterns", where="enforcement.repository_tree", lower=False + ) + try: + self.content_regex = re.compile( + "|".join(f"(?:{pattern})" for pattern in patterns), re.IGNORECASE + ) + except re.error as exc: + raise PolicyError( + f"enforcement.repository_tree.content_patterns is not valid: {exc}" + ) from exc + + # Signatures arrive as PARTS and are joined here on purpose: a whole + # banner written literally into the policy file would make this guard + # fail on its own rule file. + signature_parts = enforcement.get("content_signature_parts") + if not isinstance(signature_parts, list) or not signature_parts: + raise PolicyError( + "enforcement.content_signature_parts must be a non-empty list" + ) + signatures = [] + for entry in signature_parts: + if not isinstance(entry, list) or not entry: + raise PolicyError( + "enforcement.content_signature_parts entries must be " + "non-empty lists of parts" + ) + joined = "".join(str(part) for part in entry) + if not joined: + raise PolicyError("a content signature is empty") + signatures.append(joined) + self.content_signatures = tuple(signatures) + + repositories = document.get("public_repositories") + if not isinstance(repositories, dict) or not repositories: + raise PolicyError("public_repositories: must be a non-empty mapping") + self.public_repositories = repositories + self.policy_digest = str(document.get("policy_digest", "unknown")) + self.policy_last_updated = str(document.get("policy_last_updated", "unknown")) + + +def load_policy(path: Path = POLICY_PATH) -> SourcePolicy: + """Load the rendered policy, or raise so the caller can fail closed.""" + try: + raw = path.read_text(encoding="utf-8") + except OSError as exc: + raise PolicyError( + f"cannot read the rendered source policy {path}: {exc}. It is rendered " + "from OpenAdaptAI/openadapt-internal:source-policy.yaml and must be " + "committed in this repository" + ) from exc + try: + document = json.loads(raw) + except json.JSONDecodeError as exc: + raise PolicyError(f"{path} is not valid JSON: {exc}") from exc + if not isinstance(document, dict): + raise PolicyError(f"{path} did not parse to an object") + return SourcePolicy(document) + + def _tracked_files(root: Path) -> list[str]: result = subprocess.run( ["git", "ls-files", "-z"], @@ -134,6 +214,25 @@ def _tracked_files(root: Path) -> list[str]: return [p for p in result.stdout.decode().split("\0") if p] +def _repository_name(root: Path) -> str: + """Name of the scanned repository, used to look up its classification.""" + try: + result = subprocess.run( + ["git", "remote", "get-url", "origin"], + cwd=root, + capture_output=True, + check=True, + ) + except (subprocess.CalledProcessError, OSError): + return root.name + url = result.stdout.decode().strip() + if not url: + return root.name + # Match on the repository NAME rather than owner/name so a fork running the + # same CI is checked rather than rejected. + return url.rstrip("/").rsplit("/", 1)[-1].removesuffix(".git") or root.name + + def _is_allowlisted(path: str) -> bool: return any( path == allowed or path.startswith(allowed.rstrip("/") + "/") @@ -141,38 +240,33 @@ def _is_allowlisted(path: str) -> bool: ) -def main() -> int: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument( - "--root", - type=Path, - default=DEFAULT_ROOT, - help="Repository checkout to scan (default: this repository).", - ) - args = parser.parse_args() - root = args.root.resolve() - - try: - tracked = _tracked_files(root) - except subprocess.CalledProcessError as exc: - print(f"FATAL: git ls-files failed in {root}: {exc}", file=sys.stderr) - return 2 - - content_regex = re.compile( - "|".join(f"(?:{p})" for p in DENYLISTED_CONTENT_PATTERNS), re.IGNORECASE - ) - +def scan(root: Path, policy: SourcePolicy) -> list[str]: + """Return every boundary violation among the tracked files under `root`.""" violations: list[str] = [] - for rel_path in tracked: + for rel_path in _tracked_files(root): if _is_allowlisted(rel_path): continue lower = rel_path.lower() - for token in DENYLISTED_PATH_TOKENS: - if token in lower: + matched_token = next( + (token for token in policy.path_tokens if token in lower), None + ) + if matched_token is not None: + violations.append( + f"{rel_path}: path contains denylisted token {matched_token!r}" + ) + else: + segment = next( + ( + part + for part in lower.split("/") + if part in policy.private_path_segments + ), + None, + ) + if segment is not None: violations.append( - f"{rel_path}: path contains denylisted token {token!r}" + f"{rel_path}: path lies under private segment {segment!r}" ) - break full_path = root / rel_path if full_path.suffix.lower() in SKIP_CONTENT_SUFFIXES or not full_path.is_file(): @@ -183,13 +277,73 @@ def main() -> int: text = full_path.read_text(errors="ignore") except OSError: continue - match = content_regex.search(text) + signature = next( + (sig for sig in policy.content_signatures if sig in text), None + ) + if signature is not None: + violations.append( + f"{rel_path}: content carries the private-artifact banner" + ) + continue + match = policy.content_regex.search(text) if match: line = text.count("\n", 0, match.start()) + 1 violations.append( f"{rel_path}:{line}: content matches denylisted pattern " f"{match.group(0)!r}" ) + return violations + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--root", + type=Path, + default=DEFAULT_ROOT, + help="Repository checkout to scan (default: this repository).", + ) + parser.add_argument( + "--repo", + default=None, + help=( + "Repository name to look up in the policy (default: derived from the " + "scanned checkout's origin remote)." + ), + ) + parser.add_argument( + "--policy", + type=Path, + default=POLICY_PATH, + help="Rendered policy to enforce (default: the copy in this repository).", + ) + args = parser.parse_args() + root = args.root.resolve() + + # Fail closed: no rules, no run. + try: + policy = load_policy(args.policy) + except PolicyError as exc: + print(f"FATAL: {exc}", file=sys.stderr) + return 2 + + name = args.repo or _repository_name(root) + if name not in policy.public_repositories: + print( + f"FATAL: {name!r} is not classified as a public repository in the " + "source-availability manifest. Either it has no entry (add one to " + "OpenAdaptAI/openadapt-internal:source-policy.yaml in the same change " + "that creates the repository) or it is private, in which case this " + "public-tree guard does not apply.", + file=sys.stderr, + ) + return 2 + + try: + violations = scan(root, policy) + except subprocess.CalledProcessError as exc: + print(f"FATAL: git ls-files failed in {root}: {exc}", file=sys.stderr) + return 2 if violations: print( @@ -203,7 +357,10 @@ def main() -> int: print(f"ERROR: {violation}", file=sys.stderr) return 1 - print(f"OK: no boundary violations across {len(tracked)} tracked files.") + print( + f"OK: no boundary violations in {name} " + f"(policy {policy.policy_digest}, updated {policy.policy_last_updated})." + ) return 0 diff --git a/source-policy.public.json b/source-policy.public.json new file mode 100644 index 000000000..f7a8ddb27 --- /dev/null +++ b/source-policy.public.json @@ -0,0 +1,417 @@ +{ + "_comment": "GENERATED FILE -- DO NOT EDIT BY HAND. Rendered by OpenAdaptAI/openadapt-internal:scripts/render_public_policy.py from the canonical private manifest OpenAdaptAI/openadapt-internal:source-policy.yaml. Edit the rule there and re-render; a CI job in that private repository compares this file against the canonical manifest daily. The CI guards that read this file FAIL CLOSED when it is missing, unparseable, or incomplete.", + "crown_jewel_categories": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "enforcement": { + "built_artifacts": { + "path_prefixes": [ + "docs/validation/adversary_corpus", + "openadapt_flow/validation/adversary_corpus" + ] + }, + "content_signature_parts": [ + [ + "OPENADAPT-CORPUS", + "-PRIVATE-DO-NOT-PACKAGE" + ] + ], + "disclosed_private_repositories": [ + "openadapt-cloud", + "openadapt-corpus", + "openadapt-internal" + ], + "path_tokens": [ + "adversary_corpus", + "cerner", + "client_fixture", + "control_plane", + "customer_fixture", + "customer_recipe", + "deployment_corpus", + "deployment_thresholds", + "effect_oracle_recipe", + "enterprise_productionized", + "epic_hyperspace", + "grown_corpus", + "held_out_corpus", + "identity_roc", + "meditech", + "millennium_recipe", + "openadapt-cloud", + "openadapt-corpus", + "openadapt-internal", + "oracle_recipe", + "paid_agent_evidence", + "pixel_verify_cert", + "powerchart", + "real_emr", + "tuned_adversary" + ], + "private_path_segments": [ + ".private", + "private" + ], + "repository_tree": { + "content_patterns": [ + "powerchart", + "cerner\\s+millennium", + "epic\\s+hyperspace", + "openadapt[_-]corpus[/:]", + "deployment[_-]derived\\s+threshold\\s*=", + "oracle[_-]recipe[_-]id\\s*[:=]" + ] + } + }, + "generated_from": "OpenAdaptAI/openadapt-internal:source-policy.yaml", + "policy_digest": "sha256:09e5092a6f84cbad0b588a9aec65cc7d26492dced2ddce26ca288a3ff81992c9", + "policy_last_updated": "2026-07-28", + "policy_version": 1, + "public_repositories": { + ".github": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/.github" + }, + "OmniMCP": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/OmniMCP" + }, + "OpenAdapt": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/OpenAdapt" + }, + "OpenSanitizer": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/OpenSanitizer" + }, + "PydanticPrompt": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/PydanticPrompt" + }, + "openadapt-agent": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-agent" + }, + "openadapt-blog": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-blog" + }, + "openadapt-capture": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-capture" + }, + "openadapt-consilium": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-consilium" + }, + "openadapt-console": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-console" + }, + "openadapt-crier": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-crier" + }, + "openadapt-desktop": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-desktop" + }, + "openadapt-evals": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-evals" + }, + "openadapt-flow": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-flow" + }, + "openadapt-grounding": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-grounding" + }, + "openadapt-herald": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-herald" + }, + "openadapt-ml": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-ml" + }, + "openadapt-ops": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-ops" + }, + "openadapt-privacy": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-privacy" + }, + "openadapt-retrieval": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-retrieval" + }, + "openadapt-telemetry": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-telemetry" + }, + "openadapt-tray": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-tray" + }, + "openadapt-types": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-types" + }, + "openadapt-viewer": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-viewer" + }, + "openadapt-web": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-web" + }, + "openadapt-wright": { + "classification": "public", + "must_not_contain": [ + "control_plane", + "deployment_thresholds", + "enterprise_productionized", + "grown_corpus", + "oracle_recipes", + "real_emr_datasets", + "tuned_adversary_params" + ], + "slug": "OpenAdaptAI/openadapt-wright" + } + }, + "schema_version": 1 +} diff --git a/tests/test_source_boundary.py b/tests/test_source_boundary.py new file mode 100644 index 000000000..fcaaefc01 --- /dev/null +++ b/tests/test_source_boundary.py @@ -0,0 +1,227 @@ +"""The source-availability guard must fail closed and must never get weaker. + +The guard reads its rules from `source-policy.public.json`, which is rendered +from the canonical private manifest. Two properties matter and are pinned here: + +1. Fail closed. A missing, unparseable, incomplete, or unknown-schema policy + stops the build with exit code 2. A guard that passes with no rules is worse + than a hardcoded list, because everyone believes it ran. +2. Never weaker. `LEGACY_DENYLIST` is the hand-copied denylist this guard + carried before the rules moved into the manifest. It is a one-way ratchet: a + rule may be added, but a rule that once blocked something must keep blocking + it. Failing this test means the change removed protection. +""" + +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "scripts")) + +import check_source_boundary as guard # noqa: E402 + +# The exact denylist this script carried on 2026-07-28, before it derived its +# rules from the manifest. Ratchet only: never delete an entry to make a test +# pass. +LEGACY_DENYLIST = ( + "powerchart", + "cerner", + "millennium_recipe", + "epic_hyperspace", + "meditech", + "customer_fixture", + "customer_recipe", + "client_fixture", + "deployment_corpus", + "deployment_thresholds", + "real_emr", + "grown_corpus", + "tuned_adversary", + "adversary_corpus", + "identity_roc", + "held_out_corpus", + "oracle_recipe", + "effect_oracle_recipe", + "pixel_verify_cert", + "enterprise_productionized", + "paid_agent_evidence", + "openadapt-corpus", + "openadapt-cloud", + "openadapt-internal", +) + +LEGACY_CONTENT_PATTERNS = ( + "powerchart", + "cerner millennium", + "epic hyperspace", + "openadapt-corpus/", + "deployment-derived threshold =", + "oracle-recipe-id: 7", +) + + +@pytest.fixture() +def policy() -> guard.SourcePolicy: + return guard.load_policy() + + +def _repo(tmp_path: Path, relative: str, content: str = "placeholder\n") -> Path: + subprocess.run(["git", "init", "-q"], cwd=tmp_path, check=True) + target = tmp_path / relative + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(content, encoding="utf-8") + subprocess.run(["git", "add", relative], cwd=tmp_path, check=True) + return tmp_path + + +def _write_policy(tmp_path: Path, document: object) -> Path: + path = tmp_path / "policy.json" + path.write_text(json.dumps(document), encoding="utf-8") + return path + + +# -------------------------------------------------------------------------- +# The repository this guard defends is clean, and the guard says so. +# -------------------------------------------------------------------------- + + +def test_this_repository_passes() -> None: + assert guard.main.__module__ # import smoke + completed = subprocess.run( + [sys.executable, "scripts/check_source_boundary.py"], + cwd=ROOT, + capture_output=True, + text=True, + ) + assert completed.returncode == 0, completed.stderr + + +# -------------------------------------------------------------------------- +# Not weaker: every legacy rule still blocks. +# -------------------------------------------------------------------------- + + +@pytest.mark.parametrize("token", LEGACY_DENYLIST) +def test_legacy_path_token_still_blocked( + tmp_path: Path, policy: guard.SourcePolicy, token: str +) -> None: + root = _repo(tmp_path, f"src/{token}_fixture.json", "{}\n") + violations = guard.scan(root, policy) + assert violations, f"{token} no longer fails the path scan" + + +@pytest.mark.parametrize("phrase", LEGACY_CONTENT_PATTERNS) +def test_legacy_content_pattern_still_blocked( + tmp_path: Path, policy: guard.SourcePolicy, phrase: str +) -> None: + root = _repo(tmp_path, "docs/notes.md", f"a line mentioning {phrase} here\n") + violations = guard.scan(root, policy) + assert violations, f"content {phrase!r} no longer fails the content scan" + + +@pytest.mark.parametrize("token", LEGACY_DENYLIST) +def test_legacy_tokens_are_present_in_the_rendered_policy( + policy: guard.SourcePolicy, token: str +) -> None: + assert token in policy.path_tokens + + +def test_every_policy_token_is_actually_enforced( + tmp_path: Path, policy: guard.SourcePolicy +) -> None: + for index, token in enumerate(policy.path_tokens): + root = tmp_path / f"case{index}" + root.mkdir() + _repo(root, f"data/{token}/payload.txt") + assert guard.scan(root, policy), f"policy token {token!r} is not enforced" + + +def test_private_path_segment_is_blocked( + tmp_path: Path, policy: guard.SourcePolicy +) -> None: + root = _repo(tmp_path, "private/notes.txt") + assert guard.scan(root, policy) + + +def test_private_artifact_banner_is_blocked( + tmp_path: Path, policy: guard.SourcePolicy +) -> None: + banner = policy.content_signatures[0] + root = _repo(tmp_path, "docs/benign_name.md", f"header\n{banner}\nbody\n") + violations = guard.scan(root, policy) + assert violations and "banner" in violations[0] + + +def test_a_clean_tree_passes(tmp_path: Path, policy: guard.SourcePolicy) -> None: + root = _repo(tmp_path, "src/module.py", "def f():\n return 1\n") + assert guard.scan(root, policy) == [] + + +# -------------------------------------------------------------------------- +# Fail closed: no rules means no run. +# -------------------------------------------------------------------------- + + +def _run(argv: list[str]) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [sys.executable, "scripts/check_source_boundary.py", *argv], + cwd=ROOT, + capture_output=True, + text=True, + ) + + +def test_missing_policy_fails_closed(tmp_path: Path) -> None: + completed = _run(["--policy", str(tmp_path / "absent.json")]) + assert completed.returncode == 2 + assert "cannot read the rendered source policy" in completed.stderr + + +def test_unparseable_policy_fails_closed(tmp_path: Path) -> None: + broken = tmp_path / "policy.json" + broken.write_text("{not json", encoding="utf-8") + completed = _run(["--policy", str(broken)]) + assert completed.returncode == 2 + assert "not valid JSON" in completed.stderr + + +def test_empty_rules_fail_closed(tmp_path: Path) -> None: + document = json.loads(guard.POLICY_PATH.read_text(encoding="utf-8")) + document["enforcement"]["path_tokens"] = [] + completed = _run(["--policy", str(_write_policy(tmp_path, document))]) + assert completed.returncode == 2 + assert "path_tokens" in completed.stderr + + +def test_unknown_schema_fails_closed(tmp_path: Path) -> None: + document = json.loads(guard.POLICY_PATH.read_text(encoding="utf-8")) + document["schema_version"] = 99 + completed = _run(["--policy", str(_write_policy(tmp_path, document))]) + assert completed.returncode == 2 + assert "unknown policy schema" in completed.stderr + + +def test_missing_enforcement_block_fails_closed(tmp_path: Path) -> None: + document = json.loads(guard.POLICY_PATH.read_text(encoding="utf-8")) + del document["enforcement"] + completed = _run(["--policy", str(_write_policy(tmp_path, document))]) + assert completed.returncode == 2 + assert "enforcement" in completed.stderr + + +def test_unclassified_repository_fails_closed() -> None: + completed = _run(["--repo", "repository-with-no-manifest-entry"]) + assert completed.returncode == 2 + assert "not classified as a public repository" in completed.stderr + + +def test_this_repository_is_classified_public(policy: guard.SourcePolicy) -> None: + assert "OpenAdapt" in policy.public_repositories + assert policy.public_repositories["OpenAdapt"]["classification"] == "public" + assert policy.public_repositories["OpenAdapt"]["must_not_contain"]