#!/usr/bin/env python3 """Validate the versioned Crank Community capability-baseline snapshot.""" from __future__ import annotations import argparse import hashlib import json import re import subprocess import sys from dataclasses import dataclass from pathlib import Path from typing import Any MAX_JSON_BYTES = 4 * 1024 * 1024 MAX_CHECKLIST_BYTES = 1024 * 1024 MAX_DIAGNOSTICS = 1000 MAX_REPORT_BYTES = 65_536 VERSION_RE = re.compile(r"^[0-9]{4}\.[0-9]{2}\.[0-9]{2}\.[1-9][0-9]*$") HEX_RE = re.compile(r"^[0-9a-f]{64}$") FLOW_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$") EXPECTED_KINDS = {"inventory", "required_surfaces", "taxonomy", "checklist", "results"} IMPLEMENTATION_STATUSES = {"implemented", "planned", "gap", "blocked"} EXECUTION_VERDICTS = {"pass", "fail", "blocked", "skipped", "flaky", "not_run"} EVIDENCE_MODES = {"automated", "manual_only"} SEVERE = {"Critical", "High"} COMMAND_IDS = { "python-tooling-tests", "rust-admin-integration", "rust-mcp-integration", "ui-build", "ui-playwright", "just-verify", "authenticated-product-smoke", } CHECKLIST_VERDICTS = {"pass", "fail", "blocked", "not_run", "gap", "n/a"} CHECKLIST_STATES = {"happy", "loading", "empty", "error", "recovery", "stale", "ru-en", "safe-output"} UNSAFE_RE = re.compile(r"(?i)(bearer\s+\S+|cookie\s*[:=]|authorization\s*[:=]|https?://|/(?:home|users|root)/)") UNSUPPORTED_SCHEMA_KEYWORDS = { "allOf", "anyOf", "oneOf", "not", "if", "then", "else", "contains", "dependentSchemas", "patternProperties", "propertyNames", "unevaluatedItems", "unevaluatedProperties", "prefixItems", } @dataclass(frozen=True, order=True) class Diagnostic: code: str pointer: str class DuplicateKey(ValueError): pass def reject_duplicates(pairs: list[tuple[str, Any]]) -> dict[str, Any]: result: dict[str, Any] = {} for key, value in pairs: if key in result: raise DuplicateKey(key) result[key] = value return result def logical_path(root: Path, value: Any) -> Path | None: if not isinstance(value, str) or not value or len(value) > 1024 or any(ord(char) < 32 for char in value): return None candidate = Path(value) if candidate.is_absolute() or "\\" in value or any(part in ("", ".", "..") for part in candidate.parts): return None try: root_resolved = root.resolve(strict=True) joined = root / candidate if joined.is_symlink(): return None resolved = joined.resolve(strict=True) resolved.relative_to(root_resolved) except (OSError, ValueError): return None if not resolved.is_file(): return None return resolved def tracked_path(root: Path, value: Any) -> bool: if not isinstance(value, str): return False try: result = subprocess.run( ["git", "-C", str(root), "ls-files", "--error-unmatch", "--", value], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=False, timeout=5, ) except (OSError, subprocess.SubprocessError): return False return result.returncode == 0 def read_bytes(path: Path, limit: int) -> bytes: try: with path.open("rb") as file: raw = file.read(limit + 1) except OSError as error: raise ValueError("BROKEN") from error if len(raw) > limit: raise OverflowError return raw def read_json(path: Path) -> Any: raw = read_bytes(path, MAX_JSON_BYTES) try: return json.loads(raw.decode("utf-8"), object_pairs_hook=reject_duplicates) except (UnicodeDecodeError, json.JSONDecodeError, DuplicateKey, ValueError, RecursionError) as error: raise TypeError from error def schema_contract_valid(schema: Any) -> bool: if not isinstance(schema, dict): return False allowed_root = {"$schema", "$id", "title", "description", "type", "additionalProperties", "required", "properties", "$defs"} if set(schema) - allowed_root: return False pending = [schema] while pending: node = pending.pop() if isinstance(node, dict): if set(node) & UNSUPPORTED_SCHEMA_KEYWORDS: return False pending.extend(node.values()) elif isinstance(node, list): pending.extend(node) if schema.get("$schema") != "https://json-schema.org/draft/2020-12/schema": return False if schema.get("type") != "object" or schema.get("additionalProperties") is not False: return False if schema.get("required") != ["schema_version", "baseline_version", "artifacts"]: return False properties = schema.get("properties") definitions = schema.get("$defs") if not isinstance(properties, dict): return False if not isinstance(definitions, dict) or set(definitions) != {"artifact", "taxonomy", "run", "manual_result", "defect", "results"}: return False for name in ("artifact", "taxonomy", "run", "defect", "results"): definition = definitions.get(name) if not isinstance(definition, dict) or definition.get("type") != "object" or definition.get("additionalProperties") is not False: return False if definitions["run"].get("required") != ["id", "command_id", "flow_ids", "execution_verdict", "evidence_mode", "accepted", "source_report_sha256", "source_revision", "environment_class", "collector"]: return False if definitions["defect"].get("required") != ["id", "severity", "steps", "contract", "owner", "flow_ids", "next_action"]: return False taxonomy_properties = definitions["taxonomy"].get("properties") full_pass = taxonomy_properties.get("full_pass") if isinstance(taxonomy_properties, dict) else None if not isinstance(full_pass, dict) or set(full_pass.get("properties", {})) != {"implementation_status", "execution_verdict", "evidence_mode"}: return False manual_result = definitions.get("manual_result") if not isinstance(manual_result, dict) or manual_result.get("type") != "object" or manual_result.get("additionalProperties") is not False: return False version = properties.get("schema_version") artifacts = properties.get("artifacts") return ( isinstance(version, dict) and type(version.get("const")) is int and version.get("const") == 1 and isinstance(artifacts, dict) and artifacts.get("minItems") == 5 and artifacts.get("maxItems") == 5 ) def add(diagnostics: list[Diagnostic], code: str, pointer: str) -> None: if len(diagnostics) < MAX_DIAGNOSTICS: diagnostics.append(Diagnostic(code, pointer[:1024])) def exact_keys(value: Any, required: set[str], optional: set[str] = set()) -> bool: return isinstance(value, dict) and required <= set(value) and not (set(value) - required - optional) def nonempty_text(value: Any, limit: int) -> bool: return isinstance(value, str) and bool(value.strip()) and len(value) <= limit def parse_checklist(text: Any, diagnostics: list[Diagnostic]) -> dict[str, dict[str, Any]]: if not isinstance(text, str): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/checklist") return {} matches = list(re.finditer(r"(?m)^## (UI-[0-9]{2})\s+[^\n]+\n", text)) if not 1 <= len(matches) <= 16: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/checklist/checks") checks: dict[str, dict[str, Any]] = {} for index, match in enumerate(matches): check_id = match.group(1) block = text[match.end(): matches[index + 1].start() if index + 1 < len(matches) else len(text)] fields: dict[str, str] = {} for key in ("flow_id", "states", "verdict", "reason"): field = re.search(rf"(?m)^- {key}:\s*(.+?)\s*$", block) if field: fields[key] = field.group(1) if check_id in checks or set(fields) != {"flow_id", "states", "verdict", "reason"}: add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/checklist/{check_id}") continue states = [item.strip() for item in fields["states"].split(",") if item.strip()] if not states or len(states) != len(set(states)) or set(states) - CHECKLIST_STATES: add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/checklist/{check_id}/states") if fields["verdict"] not in CHECKLIST_VERDICTS or not nonempty_text(fields["reason"], 2048): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/checklist/{check_id}/verdict") checks[check_id] = {"flow_id": fields["flow_id"], "verdict": fields["verdict"]} return checks def validate(root: Path, manifest_path: str, schema_path: str) -> tuple[list[Diagnostic], str | None, int]: diagnostics: list[Diagnostic] = [] manifest_file = logical_path(root, manifest_path) schema_file = logical_path(root, schema_path) if manifest_file is None: add(diagnostics, "BROKEN_ARTIFACT_LINK", "/manifest") if schema_file is None: add(diagnostics, "BROKEN_ARTIFACT_LINK", "/schema") if diagnostics: return diagnostics, None, 0 try: manifest = read_json(manifest_file) # type: ignore[arg-type] schema = read_json(schema_file) # type: ignore[arg-type] except OverflowError: add(diagnostics, "INPUT_TOO_LARGE", "/") return diagnostics, None, 0 except TypeError: add(diagnostics, "INVALID_JSON", "/") return diagnostics, None, 0 except ValueError: add(diagnostics, "BROKEN_ARTIFACT_LINK", "/") return diagnostics, None, 0 if not schema_contract_valid(schema): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/schema") if not isinstance(manifest, dict): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/manifest") return diagnostics, None, 0 if set(manifest) != {"schema_version", "baseline_version", "artifacts"}: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/manifest") version = manifest.get("baseline_version") if not isinstance(version, str) or not VERSION_RE.fullmatch(version): add(diagnostics, "BASELINE_VERSION_MISMATCH", "/baseline_version") version = None if type(manifest.get("schema_version")) is not int or manifest.get("schema_version") != 1: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/schema_version") artifacts = manifest.get("artifacts") if not isinstance(artifacts, list) or len(artifacts) != 5: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/artifacts") return diagnostics, version, 0 loaded: dict[str, Any] = {} seen: set[str] = set() for index, item in enumerate(artifacts): pointer = f"/artifacts/{index}" if not isinstance(item, dict): add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer) continue if set(item) != {"kind", "path", "sha256"}: add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer) kind = item.get("kind") if kind not in EXPECTED_KINDS or kind in seen: add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/kind") continue seen.add(kind) artifact_file = logical_path(root, item.get("path")) if artifact_file is None: add(diagnostics, "BROKEN_ARTIFACT_LINK", f"{pointer}/path") continue if not tracked_path(root, item.get("path")): add(diagnostics, "UNTRACKED_ARTIFACT", f"{pointer}/path") expected_hash = item.get("sha256") if not isinstance(expected_hash, str) or not HEX_RE.fullmatch(expected_hash): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/sha256") continue try: raw = read_bytes(artifact_file, MAX_CHECKLIST_BYTES if kind == "checklist" else MAX_JSON_BYTES) except OverflowError: add(diagnostics, "INPUT_TOO_LARGE", pointer) continue except ValueError: add(diagnostics, "BROKEN_ARTIFACT_LINK", pointer) continue if hashlib.sha256(raw).hexdigest() != expected_hash: add(diagnostics, "CHECKSUM_MISMATCH", f"{pointer}/sha256") if kind == "checklist": try: text = raw.decode("utf-8") except UnicodeDecodeError: add(diagnostics, "INVALID_JSON", pointer) continue match = re.search(r"(?m)^baseline_version:\s*([^\s]+)\s*$", text) if not match or match.group(1) != version: add(diagnostics, "BASELINE_VERSION_MISMATCH", pointer) loaded[kind] = text else: try: loaded[kind] = json.loads(raw.decode("utf-8"), object_pairs_hook=reject_duplicates) except (UnicodeDecodeError, json.JSONDecodeError, DuplicateKey, ValueError, RecursionError): add(diagnostics, "INVALID_JSON", pointer) if seen != EXPECTED_KINDS: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/artifacts") checklist_checks = parse_checklist(loaded.get("checklist"), diagnostics) validate_loaded(loaded, version, checklist_checks, diagnostics) return diagnostics, version, len(loaded.get("inventory", {}).get("flows", [])) if isinstance(loaded.get("inventory"), dict) else 0 def validate_loaded( loaded: dict[str, Any], version: str | None, checklist_checks: dict[str, dict[str, Any]], diagnostics: list[Diagnostic], ) -> None: for kind in ("required_surfaces", "taxonomy", "results"): value = loaded.get(kind) if not isinstance(value, dict) or value.get("baseline_version") != version: add(diagnostics, "BASELINE_VERSION_MISMATCH", f"/{kind}/baseline_version") inventory = loaded.get("inventory") flows = inventory.get("flows") if isinstance(inventory, dict) else None if not isinstance(flows, list): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/inventory/flows") return statuses: dict[str, str] = {} for index, flow in enumerate(flows): if not isinstance(flow, dict) or not FLOW_RE.fullmatch(str(flow.get("id", ""))): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/inventory/flows/{index}") continue flow_id = flow["id"] status = flow.get("status") if flow_id in statuses or status not in IMPLEMENTATION_STATUSES: add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/inventory/flows/{index}") continue statuses[flow_id] = status required = loaded.get("required_surfaces") required_ids = required.get("required_flow_ids") if exact_keys(required, {"baseline_version", "required_flow_ids", "surface_groups"}) else None current_ids = {flow_id for flow_id, status in statuses.items() if status != "planned"} if ( not isinstance(required_ids, list) or any(not isinstance(flow_id, str) for flow_id in required_ids) or len(required_ids) != len(set(required_ids)) or set(required_ids) != current_ids ): add(diagnostics, "INVALID_REQUIRED_SURFACES", "/required_surfaces/required_flow_ids") required_ids = [] groups = required.get("surface_groups") if isinstance(required, dict) else None grouped: list[str] = [] group_ids: set[str] = set() if not isinstance(groups, list) or not 1 <= len(groups) <= 64: add(diagnostics, "INVALID_REQUIRED_SURFACES", "/required_surfaces/surface_groups") groups = [] for index, group in enumerate(groups): pointer = f"/required_surfaces/surface_groups/{index}" if not exact_keys(group, {"id", "flow_ids"}) or not FLOW_RE.fullmatch(str(group.get("id", ""))): add(diagnostics, "INVALID_REQUIRED_SURFACES", pointer) continue group_flow_ids = group.get("flow_ids") if group["id"] in group_ids or not isinstance(group_flow_ids, list) or not group_flow_ids or any(not isinstance(item, str) for item in group_flow_ids): add(diagnostics, "INVALID_REQUIRED_SURFACES", pointer) continue group_ids.add(group["id"]) grouped.extend(group_flow_ids) if len(grouped) != len(set(grouped)) or set(grouped) != set(required_ids): add(diagnostics, "INVALID_REQUIRED_SURFACES", "/required_surfaces/surface_groups") taxonomy = loaded.get("taxonomy") if isinstance(taxonomy, dict): taxonomy_required = {"baseline_version", "implementation_statuses", "execution_verdicts", "evidence_modes", "full_pass"} taxonomy_optional = {"manual_only_rule", "non_pass_rule"} if not exact_keys(taxonomy, taxonomy_required, taxonomy_optional): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy") for rule in taxonomy_optional & set(taxonomy): if not nonempty_text(taxonomy.get(rule), 2048): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/taxonomy/{rule}") if set(taxonomy.get("implementation_statuses", [])) != IMPLEMENTATION_STATUSES: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy/implementation_statuses") if set(taxonomy.get("execution_verdicts", [])) != EXECUTION_VERDICTS: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy/execution_verdicts") if set(taxonomy.get("evidence_modes", [])) != EVIDENCE_MODES: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy/evidence_modes") if taxonomy.get("full_pass") != {"implementation_status": "implemented", "execution_verdict": "pass", "evidence_mode": "automated"}: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy/full_pass") results = loaded.get("results") if not isinstance(results, dict): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results") return if set(results) != {"baseline_version", "source_revision", "environment_class", "runs", "manual_results", "defects"}: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results") root_revision = results.get("source_revision") root_environment = results.get("environment_class") if not isinstance(root_revision, str) or not re.fullmatch(r"[0-9a-f]{40}", root_revision): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/source_revision") if not isinstance(root_environment, str) or not re.fullmatch(r"[a-z0-9-]{1,64}", root_environment): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/environment_class") safe_serialized = json.dumps(results, sort_keys=True) if UNSAFE_RE.search(safe_serialized): add(diagnostics, "UNSAFE_EVIDENCE", "/results") evidenced: set[str] = set() seen_run_ids: set[str] = set() seen_hashes: set[str] = set() runs = results.get("runs") if not isinstance(runs, list): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/runs") runs = [] for index, run in enumerate(runs): pointer = f"/results/runs/{index}" if not isinstance(run, dict): add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer) continue required_run_keys = {"id", "command_id", "flow_ids", "execution_verdict", "evidence_mode", "accepted", "source_report_sha256", "source_revision", "environment_class", "collector"} if not exact_keys(run, required_run_keys, {"summary", "safe_outcome"}): add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer) verdict = run.get("execution_verdict") mode = run.get("evidence_mode") accepted = run.get("accepted") source_revision = run.get("source_revision") environment_class = run.get("environment_class") command_id = run.get("command_id") source_hash = run.get("source_report_sha256") run_id = run.get("id") if not nonempty_text(run_id, 128) or run_id in seen_run_ids: add(diagnostics, "DUPLICATE_RUN_ID", f"{pointer}/id") elif isinstance(run_id, str): seen_run_ids.add(run_id) if command_id not in COMMAND_IDS: add(diagnostics, "UNKNOWN_COMMAND", f"{pointer}/command_id") if not isinstance(source_revision, str) or not re.fullmatch(r"[0-9a-f]{40}", source_revision): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/source_revision") elif source_revision != root_revision: add(diagnostics, "PROVENANCE_MISMATCH", f"{pointer}/source_revision") if not isinstance(environment_class, str) or not re.fullmatch(r"[a-z0-9-]{1,64}", environment_class): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/environment_class") elif environment_class != root_environment: add(diagnostics, "PROVENANCE_MISMATCH", f"{pointer}/environment_class") if not isinstance(source_hash, str) or not HEX_RE.fullmatch(source_hash): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/source_report_sha256") elif source_hash in seen_hashes: add(diagnostics, "DUPLICATE_REPORT_HASH", f"{pointer}/source_report_sha256") else: seen_hashes.add(source_hash) if verdict not in EXECUTION_VERDICTS or mode not in EVIDENCE_MODES: add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer) expected_accepted = verdict == "pass" and mode == "automated" and run.get("collector") == "capability-baseline-collector-v1" if type(accepted) is not bool or accepted is not expected_accepted: add(diagnostics, "NON_PASS_RECORDED_AS_PASS", pointer) flow_ids = run.get("flow_ids") if not isinstance(flow_ids, list) or not flow_ids or any(not isinstance(flow_id, str) for flow_id in flow_ids) or len(flow_ids) != len(set(flow_ids)): add(diagnostics, "MISSING_FLOW_EVIDENCE", f"{pointer}/flow_ids") continue for flow_id in flow_ids: if flow_id not in statuses: add(diagnostics, "UNKNOWN_FLOW_ID", f"{pointer}/flow_ids") elif accepted is True and statuses[flow_id] != "implemented": add(diagnostics, "NON_PASS_RECORDED_AS_PASS", f"{pointer}/flow_ids") elif accepted is True: evidenced.add(flow_id) for flow_id, status in statuses.items(): if status == "implemented" and flow_id not in evidenced: add(diagnostics, "MISSING_FLOW_EVIDENCE", f"/inventory/{flow_id}") manual_results = results.get("manual_results") if not isinstance(manual_results, list) or len(manual_results) > 1000: add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/manual_results") manual_results = [] seen_checks: set[str] = set() for index, manual in enumerate(manual_results): pointer = f"/results/manual_results/{index}" required_manual_keys = {"check_id", "evidence_mode", "execution_verdict", "flow_ids", "next_evidence"} if not exact_keys(manual, required_manual_keys): add(diagnostics, "INVALID_MANUAL_EVIDENCE", pointer) continue check_id = manual.get("check_id") flow_ids = manual.get("flow_ids") if check_id in seen_checks or check_id not in checklist_checks: add(diagnostics, "INVALID_MANUAL_EVIDENCE", f"{pointer}/check_id") elif isinstance(check_id, str): seen_checks.add(check_id) if manual.get("evidence_mode") != "manual_only" or manual.get("execution_verdict") not in EXECUTION_VERDICTS: add(diagnostics, "INVALID_MANUAL_EVIDENCE", pointer) if not nonempty_text(manual.get("next_evidence"), 2048): add(diagnostics, "INVALID_MANUAL_EVIDENCE", f"{pointer}/next_evidence") if not isinstance(flow_ids, list) or not flow_ids or any(not isinstance(flow_id, str) or flow_id not in statuses for flow_id in flow_ids): add(diagnostics, "INVALID_MANUAL_EVIDENCE", f"{pointer}/flow_ids") elif isinstance(check_id, str) and check_id in checklist_checks: if checklist_checks[check_id]["flow_id"] not in flow_ids or checklist_checks[check_id]["verdict"] != manual.get("execution_verdict"): add(diagnostics, "INVALID_MANUAL_EVIDENCE", pointer) if seen_checks != set(checklist_checks): add(diagnostics, "INVALID_MANUAL_EVIDENCE", "/results/manual_results") defects = results.get("defects") if not isinstance(defects, list): add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/defects") defects = [] for index, defect in enumerate(defects): if not isinstance(defect, dict): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}") continue required_fields = ("id", "severity", "contract", "owner", "steps", "next_action", "flow_ids") if set(defect) != set(required_fields): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}") continue if defect.get("severity") not in {"Critical", "High", "Medium", "Low"}: add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}/severity") if not isinstance(defect.get("steps"), list) or not 1 <= len(defect["steps"]) <= 20: add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}/steps") defect_flow_ids = defect.get("flow_ids") if not isinstance(defect_flow_ids, list) or not defect_flow_ids or any(not isinstance(flow_id, str) for flow_id in defect_flow_ids): add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}/flow_ids") defect_flow_ids = [] for flow_id in defect_flow_ids: if flow_id not in statuses: add(diagnostics, "UNKNOWN_FLOW_ID", f"/results/defects/{index}/flow_ids") elif defect.get("severity") in SEVERE and statuses[flow_id] != "blocked": add(diagnostics, "SEVERE_DEFECT_FLOW_NOT_BLOCKED", f"/results/defects/{index}") def render(diagnostics: list[Diagnostic]) -> str: lines = [f"{item.code} pointer={item.pointer}" for item in sorted(set(diagnostics))] encoded = "\n".join(lines) + ("\n" if lines else "") raw = encoded.encode("utf-8") if len(raw) <= MAX_REPORT_BYTES: return encoded marker = "REPORT_TRUNCATED pointer=/\n" budget = MAX_REPORT_BYTES - len(marker.encode("utf-8")) return raw[:budget].decode("utf-8", "ignore") + marker def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--root", default=".") parser.add_argument("--manifest", required=True) parser.add_argument("--schema", required=True) args = parser.parse_args() root = Path(args.root) try: diagnostics, version, flow_count = validate(root, args.manifest, args.schema) except (OSError, ValueError, RecursionError): diagnostics, version, flow_count = [Diagnostic("VALIDATION_FAILED", "/")], None, 0 if diagnostics: sys.stderr.write(render(diagnostics)) return 1 print(f"Capability baseline validation passed: baseline_version={version} flows={flow_count}") return 0 if __name__ == "__main__": raise SystemExit(main())