feat: harden community production foundation through story 1.5
This commit is contained in:
@@ -0,0 +1,542 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Validate the versioned Crank Community capability-baseline snapshot."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
MAX_JSON_BYTES = 4 * 1024 * 1024
|
||||
MAX_CHECKLIST_BYTES = 1024 * 1024
|
||||
MAX_DIAGNOSTICS = 1000
|
||||
MAX_REPORT_BYTES = 65_536
|
||||
VERSION_RE = re.compile(r"^[0-9]{4}\.[0-9]{2}\.[0-9]{2}\.[1-9][0-9]*$")
|
||||
HEX_RE = re.compile(r"^[0-9a-f]{64}$")
|
||||
FLOW_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
|
||||
EXPECTED_KINDS = {"inventory", "required_surfaces", "taxonomy", "checklist", "results"}
|
||||
IMPLEMENTATION_STATUSES = {"implemented", "planned", "gap", "blocked"}
|
||||
EXECUTION_VERDICTS = {"pass", "fail", "blocked", "skipped", "flaky", "not_run"}
|
||||
EVIDENCE_MODES = {"automated", "manual_only"}
|
||||
SEVERE = {"Critical", "High"}
|
||||
COMMAND_IDS = {
|
||||
"python-tooling-tests", "rust-admin-integration", "rust-mcp-integration",
|
||||
"ui-build", "ui-playwright", "just-verify", "authenticated-product-smoke",
|
||||
}
|
||||
CHECKLIST_VERDICTS = {"pass", "fail", "blocked", "not_run", "gap", "n/a"}
|
||||
CHECKLIST_STATES = {"happy", "loading", "empty", "error", "recovery", "stale", "ru-en", "safe-output"}
|
||||
UNSAFE_RE = re.compile(r"(?i)(bearer\s+\S+|cookie\s*[:=]|authorization\s*[:=]|https?://|/(?:home|users|root)/)")
|
||||
UNSUPPORTED_SCHEMA_KEYWORDS = {
|
||||
"allOf", "anyOf", "oneOf", "not", "if", "then", "else", "contains",
|
||||
"dependentSchemas", "patternProperties", "propertyNames", "unevaluatedItems",
|
||||
"unevaluatedProperties", "prefixItems",
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True, order=True)
|
||||
class Diagnostic:
|
||||
code: str
|
||||
pointer: str
|
||||
|
||||
|
||||
class DuplicateKey(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def reject_duplicates(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
||||
result: dict[str, Any] = {}
|
||||
for key, value in pairs:
|
||||
if key in result:
|
||||
raise DuplicateKey(key)
|
||||
result[key] = value
|
||||
return result
|
||||
|
||||
|
||||
def logical_path(root: Path, value: Any) -> Path | None:
|
||||
if not isinstance(value, str) or not value or len(value) > 1024 or any(ord(char) < 32 for char in value):
|
||||
return None
|
||||
candidate = Path(value)
|
||||
if candidate.is_absolute() or "\\" in value or any(part in ("", ".", "..") for part in candidate.parts):
|
||||
return None
|
||||
try:
|
||||
root_resolved = root.resolve(strict=True)
|
||||
joined = root / candidate
|
||||
if joined.is_symlink():
|
||||
return None
|
||||
resolved = joined.resolve(strict=True)
|
||||
resolved.relative_to(root_resolved)
|
||||
except (OSError, ValueError):
|
||||
return None
|
||||
if not resolved.is_file():
|
||||
return None
|
||||
return resolved
|
||||
|
||||
|
||||
def tracked_path(root: Path, value: Any) -> bool:
|
||||
if not isinstance(value, str):
|
||||
return False
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["git", "-C", str(root), "ls-files", "--error-unmatch", "--", value],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
check=False,
|
||||
timeout=5,
|
||||
)
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
return False
|
||||
return result.returncode == 0
|
||||
|
||||
|
||||
def read_bytes(path: Path, limit: int) -> bytes:
|
||||
try:
|
||||
with path.open("rb") as file:
|
||||
raw = file.read(limit + 1)
|
||||
except OSError as error:
|
||||
raise ValueError("BROKEN") from error
|
||||
if len(raw) > limit:
|
||||
raise OverflowError
|
||||
return raw
|
||||
|
||||
|
||||
def read_json(path: Path) -> Any:
|
||||
raw = read_bytes(path, MAX_JSON_BYTES)
|
||||
try:
|
||||
return json.loads(raw.decode("utf-8"), object_pairs_hook=reject_duplicates)
|
||||
except (UnicodeDecodeError, json.JSONDecodeError, DuplicateKey, ValueError, RecursionError) as error:
|
||||
raise TypeError from error
|
||||
|
||||
|
||||
def schema_contract_valid(schema: Any) -> bool:
|
||||
if not isinstance(schema, dict):
|
||||
return False
|
||||
allowed_root = {"$schema", "$id", "title", "description", "type", "additionalProperties", "required", "properties", "$defs"}
|
||||
if set(schema) - allowed_root:
|
||||
return False
|
||||
pending = [schema]
|
||||
while pending:
|
||||
node = pending.pop()
|
||||
if isinstance(node, dict):
|
||||
if set(node) & UNSUPPORTED_SCHEMA_KEYWORDS:
|
||||
return False
|
||||
pending.extend(node.values())
|
||||
elif isinstance(node, list):
|
||||
pending.extend(node)
|
||||
if schema.get("$schema") != "https://json-schema.org/draft/2020-12/schema":
|
||||
return False
|
||||
if schema.get("type") != "object" or schema.get("additionalProperties") is not False:
|
||||
return False
|
||||
if schema.get("required") != ["schema_version", "baseline_version", "artifacts"]:
|
||||
return False
|
||||
properties = schema.get("properties")
|
||||
definitions = schema.get("$defs")
|
||||
if not isinstance(properties, dict):
|
||||
return False
|
||||
if not isinstance(definitions, dict) or set(definitions) != {"artifact", "taxonomy", "run", "manual_result", "defect", "results"}:
|
||||
return False
|
||||
for name in ("artifact", "taxonomy", "run", "defect", "results"):
|
||||
definition = definitions.get(name)
|
||||
if not isinstance(definition, dict) or definition.get("type") != "object" or definition.get("additionalProperties") is not False:
|
||||
return False
|
||||
if definitions["run"].get("required") != ["id", "command_id", "flow_ids", "execution_verdict", "evidence_mode", "accepted", "source_report_sha256", "source_revision", "environment_class", "collector"]:
|
||||
return False
|
||||
if definitions["defect"].get("required") != ["id", "severity", "steps", "contract", "owner", "flow_ids", "next_action"]:
|
||||
return False
|
||||
taxonomy_properties = definitions["taxonomy"].get("properties")
|
||||
full_pass = taxonomy_properties.get("full_pass") if isinstance(taxonomy_properties, dict) else None
|
||||
if not isinstance(full_pass, dict) or set(full_pass.get("properties", {})) != {"implementation_status", "execution_verdict", "evidence_mode"}:
|
||||
return False
|
||||
manual_result = definitions.get("manual_result")
|
||||
if not isinstance(manual_result, dict) or manual_result.get("type") != "object" or manual_result.get("additionalProperties") is not False:
|
||||
return False
|
||||
version = properties.get("schema_version")
|
||||
artifacts = properties.get("artifacts")
|
||||
return (
|
||||
isinstance(version, dict)
|
||||
and type(version.get("const")) is int
|
||||
and version.get("const") == 1
|
||||
and isinstance(artifacts, dict)
|
||||
and artifacts.get("minItems") == 5
|
||||
and artifacts.get("maxItems") == 5
|
||||
)
|
||||
|
||||
|
||||
def add(diagnostics: list[Diagnostic], code: str, pointer: str) -> None:
|
||||
if len(diagnostics) < MAX_DIAGNOSTICS:
|
||||
diagnostics.append(Diagnostic(code, pointer[:1024]))
|
||||
|
||||
|
||||
def exact_keys(value: Any, required: set[str], optional: set[str] = set()) -> bool:
|
||||
return isinstance(value, dict) and required <= set(value) and not (set(value) - required - optional)
|
||||
|
||||
|
||||
def nonempty_text(value: Any, limit: int) -> bool:
|
||||
return isinstance(value, str) and bool(value.strip()) and len(value) <= limit
|
||||
|
||||
|
||||
def parse_checklist(text: Any, diagnostics: list[Diagnostic]) -> dict[str, dict[str, Any]]:
|
||||
if not isinstance(text, str):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/checklist")
|
||||
return {}
|
||||
matches = list(re.finditer(r"(?m)^## (UI-[0-9]{2})\s+[^\n]+\n", text))
|
||||
if not 1 <= len(matches) <= 16:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/checklist/checks")
|
||||
checks: dict[str, dict[str, Any]] = {}
|
||||
for index, match in enumerate(matches):
|
||||
check_id = match.group(1)
|
||||
block = text[match.end(): matches[index + 1].start() if index + 1 < len(matches) else len(text)]
|
||||
fields: dict[str, str] = {}
|
||||
for key in ("flow_id", "states", "verdict", "reason"):
|
||||
field = re.search(rf"(?m)^- {key}:\s*(.+?)\s*$", block)
|
||||
if field:
|
||||
fields[key] = field.group(1)
|
||||
if check_id in checks or set(fields) != {"flow_id", "states", "verdict", "reason"}:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/checklist/{check_id}")
|
||||
continue
|
||||
states = [item.strip() for item in fields["states"].split(",") if item.strip()]
|
||||
if not states or len(states) != len(set(states)) or set(states) - CHECKLIST_STATES:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/checklist/{check_id}/states")
|
||||
if fields["verdict"] not in CHECKLIST_VERDICTS or not nonempty_text(fields["reason"], 2048):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/checklist/{check_id}/verdict")
|
||||
checks[check_id] = {"flow_id": fields["flow_id"], "verdict": fields["verdict"]}
|
||||
return checks
|
||||
|
||||
|
||||
def validate(root: Path, manifest_path: str, schema_path: str) -> tuple[list[Diagnostic], str | None, int]:
|
||||
diagnostics: list[Diagnostic] = []
|
||||
manifest_file = logical_path(root, manifest_path)
|
||||
schema_file = logical_path(root, schema_path)
|
||||
if manifest_file is None:
|
||||
add(diagnostics, "BROKEN_ARTIFACT_LINK", "/manifest")
|
||||
if schema_file is None:
|
||||
add(diagnostics, "BROKEN_ARTIFACT_LINK", "/schema")
|
||||
if diagnostics:
|
||||
return diagnostics, None, 0
|
||||
try:
|
||||
manifest = read_json(manifest_file) # type: ignore[arg-type]
|
||||
schema = read_json(schema_file) # type: ignore[arg-type]
|
||||
except OverflowError:
|
||||
add(diagnostics, "INPUT_TOO_LARGE", "/")
|
||||
return diagnostics, None, 0
|
||||
except TypeError:
|
||||
add(diagnostics, "INVALID_JSON", "/")
|
||||
return diagnostics, None, 0
|
||||
except ValueError:
|
||||
add(diagnostics, "BROKEN_ARTIFACT_LINK", "/")
|
||||
return diagnostics, None, 0
|
||||
if not schema_contract_valid(schema):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/schema")
|
||||
if not isinstance(manifest, dict):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/manifest")
|
||||
return diagnostics, None, 0
|
||||
if set(manifest) != {"schema_version", "baseline_version", "artifacts"}:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/manifest")
|
||||
version = manifest.get("baseline_version")
|
||||
if not isinstance(version, str) or not VERSION_RE.fullmatch(version):
|
||||
add(diagnostics, "BASELINE_VERSION_MISMATCH", "/baseline_version")
|
||||
version = None
|
||||
if type(manifest.get("schema_version")) is not int or manifest.get("schema_version") != 1:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/schema_version")
|
||||
artifacts = manifest.get("artifacts")
|
||||
if not isinstance(artifacts, list) or len(artifacts) != 5:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/artifacts")
|
||||
return diagnostics, version, 0
|
||||
loaded: dict[str, Any] = {}
|
||||
seen: set[str] = set()
|
||||
for index, item in enumerate(artifacts):
|
||||
pointer = f"/artifacts/{index}"
|
||||
if not isinstance(item, dict):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer)
|
||||
continue
|
||||
if set(item) != {"kind", "path", "sha256"}:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer)
|
||||
kind = item.get("kind")
|
||||
if kind not in EXPECTED_KINDS or kind in seen:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/kind")
|
||||
continue
|
||||
seen.add(kind)
|
||||
artifact_file = logical_path(root, item.get("path"))
|
||||
if artifact_file is None:
|
||||
add(diagnostics, "BROKEN_ARTIFACT_LINK", f"{pointer}/path")
|
||||
continue
|
||||
if not tracked_path(root, item.get("path")):
|
||||
add(diagnostics, "UNTRACKED_ARTIFACT", f"{pointer}/path")
|
||||
expected_hash = item.get("sha256")
|
||||
if not isinstance(expected_hash, str) or not HEX_RE.fullmatch(expected_hash):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/sha256")
|
||||
continue
|
||||
try:
|
||||
raw = read_bytes(artifact_file, MAX_CHECKLIST_BYTES if kind == "checklist" else MAX_JSON_BYTES)
|
||||
except OverflowError:
|
||||
add(diagnostics, "INPUT_TOO_LARGE", pointer)
|
||||
continue
|
||||
except ValueError:
|
||||
add(diagnostics, "BROKEN_ARTIFACT_LINK", pointer)
|
||||
continue
|
||||
if hashlib.sha256(raw).hexdigest() != expected_hash:
|
||||
add(diagnostics, "CHECKSUM_MISMATCH", f"{pointer}/sha256")
|
||||
if kind == "checklist":
|
||||
try:
|
||||
text = raw.decode("utf-8")
|
||||
except UnicodeDecodeError:
|
||||
add(diagnostics, "INVALID_JSON", pointer)
|
||||
continue
|
||||
match = re.search(r"(?m)^baseline_version:\s*([^\s]+)\s*$", text)
|
||||
if not match or match.group(1) != version:
|
||||
add(diagnostics, "BASELINE_VERSION_MISMATCH", pointer)
|
||||
loaded[kind] = text
|
||||
else:
|
||||
try:
|
||||
loaded[kind] = json.loads(raw.decode("utf-8"), object_pairs_hook=reject_duplicates)
|
||||
except (UnicodeDecodeError, json.JSONDecodeError, DuplicateKey, ValueError, RecursionError):
|
||||
add(diagnostics, "INVALID_JSON", pointer)
|
||||
if seen != EXPECTED_KINDS:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/artifacts")
|
||||
checklist_checks = parse_checklist(loaded.get("checklist"), diagnostics)
|
||||
validate_loaded(loaded, version, checklist_checks, diagnostics)
|
||||
return diagnostics, version, len(loaded.get("inventory", {}).get("flows", [])) if isinstance(loaded.get("inventory"), dict) else 0
|
||||
|
||||
|
||||
def validate_loaded(
|
||||
loaded: dict[str, Any],
|
||||
version: str | None,
|
||||
checklist_checks: dict[str, dict[str, Any]],
|
||||
diagnostics: list[Diagnostic],
|
||||
) -> None:
|
||||
for kind in ("required_surfaces", "taxonomy", "results"):
|
||||
value = loaded.get(kind)
|
||||
if not isinstance(value, dict) or value.get("baseline_version") != version:
|
||||
add(diagnostics, "BASELINE_VERSION_MISMATCH", f"/{kind}/baseline_version")
|
||||
inventory = loaded.get("inventory")
|
||||
flows = inventory.get("flows") if isinstance(inventory, dict) else None
|
||||
if not isinstance(flows, list):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/inventory/flows")
|
||||
return
|
||||
statuses: dict[str, str] = {}
|
||||
for index, flow in enumerate(flows):
|
||||
if not isinstance(flow, dict) or not FLOW_RE.fullmatch(str(flow.get("id", ""))):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/inventory/flows/{index}")
|
||||
continue
|
||||
flow_id = flow["id"]
|
||||
status = flow.get("status")
|
||||
if flow_id in statuses or status not in IMPLEMENTATION_STATUSES:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/inventory/flows/{index}")
|
||||
continue
|
||||
statuses[flow_id] = status
|
||||
required = loaded.get("required_surfaces")
|
||||
required_ids = required.get("required_flow_ids") if exact_keys(required, {"baseline_version", "required_flow_ids", "surface_groups"}) else None
|
||||
current_ids = {flow_id for flow_id, status in statuses.items() if status != "planned"}
|
||||
if (
|
||||
not isinstance(required_ids, list)
|
||||
or any(not isinstance(flow_id, str) for flow_id in required_ids)
|
||||
or len(required_ids) != len(set(required_ids))
|
||||
or set(required_ids) != current_ids
|
||||
):
|
||||
add(diagnostics, "INVALID_REQUIRED_SURFACES", "/required_surfaces/required_flow_ids")
|
||||
required_ids = []
|
||||
groups = required.get("surface_groups") if isinstance(required, dict) else None
|
||||
grouped: list[str] = []
|
||||
group_ids: set[str] = set()
|
||||
if not isinstance(groups, list) or not 1 <= len(groups) <= 64:
|
||||
add(diagnostics, "INVALID_REQUIRED_SURFACES", "/required_surfaces/surface_groups")
|
||||
groups = []
|
||||
for index, group in enumerate(groups):
|
||||
pointer = f"/required_surfaces/surface_groups/{index}"
|
||||
if not exact_keys(group, {"id", "flow_ids"}) or not FLOW_RE.fullmatch(str(group.get("id", ""))):
|
||||
add(diagnostics, "INVALID_REQUIRED_SURFACES", pointer)
|
||||
continue
|
||||
group_flow_ids = group.get("flow_ids")
|
||||
if group["id"] in group_ids or not isinstance(group_flow_ids, list) or not group_flow_ids or any(not isinstance(item, str) for item in group_flow_ids):
|
||||
add(diagnostics, "INVALID_REQUIRED_SURFACES", pointer)
|
||||
continue
|
||||
group_ids.add(group["id"])
|
||||
grouped.extend(group_flow_ids)
|
||||
if len(grouped) != len(set(grouped)) or set(grouped) != set(required_ids):
|
||||
add(diagnostics, "INVALID_REQUIRED_SURFACES", "/required_surfaces/surface_groups")
|
||||
taxonomy = loaded.get("taxonomy")
|
||||
if isinstance(taxonomy, dict):
|
||||
taxonomy_required = {"baseline_version", "implementation_statuses", "execution_verdicts", "evidence_modes", "full_pass"}
|
||||
taxonomy_optional = {"manual_only_rule", "non_pass_rule"}
|
||||
if not exact_keys(taxonomy, taxonomy_required, taxonomy_optional):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy")
|
||||
for rule in taxonomy_optional & set(taxonomy):
|
||||
if not nonempty_text(taxonomy.get(rule), 2048):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/taxonomy/{rule}")
|
||||
if set(taxonomy.get("implementation_statuses", [])) != IMPLEMENTATION_STATUSES:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy/implementation_statuses")
|
||||
if set(taxonomy.get("execution_verdicts", [])) != EXECUTION_VERDICTS:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy/execution_verdicts")
|
||||
if set(taxonomy.get("evidence_modes", [])) != EVIDENCE_MODES:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy/evidence_modes")
|
||||
if taxonomy.get("full_pass") != {"implementation_status": "implemented", "execution_verdict": "pass", "evidence_mode": "automated"}:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/taxonomy/full_pass")
|
||||
results = loaded.get("results")
|
||||
if not isinstance(results, dict):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results")
|
||||
return
|
||||
if set(results) != {"baseline_version", "source_revision", "environment_class", "runs", "manual_results", "defects"}:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results")
|
||||
root_revision = results.get("source_revision")
|
||||
root_environment = results.get("environment_class")
|
||||
if not isinstance(root_revision, str) or not re.fullmatch(r"[0-9a-f]{40}", root_revision):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/source_revision")
|
||||
if not isinstance(root_environment, str) or not re.fullmatch(r"[a-z0-9-]{1,64}", root_environment):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/environment_class")
|
||||
safe_serialized = json.dumps(results, sort_keys=True)
|
||||
if UNSAFE_RE.search(safe_serialized):
|
||||
add(diagnostics, "UNSAFE_EVIDENCE", "/results")
|
||||
evidenced: set[str] = set()
|
||||
seen_run_ids: set[str] = set()
|
||||
seen_hashes: set[str] = set()
|
||||
runs = results.get("runs")
|
||||
if not isinstance(runs, list):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/runs")
|
||||
runs = []
|
||||
for index, run in enumerate(runs):
|
||||
pointer = f"/results/runs/{index}"
|
||||
if not isinstance(run, dict):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer)
|
||||
continue
|
||||
required_run_keys = {"id", "command_id", "flow_ids", "execution_verdict", "evidence_mode", "accepted", "source_report_sha256", "source_revision", "environment_class", "collector"}
|
||||
if not exact_keys(run, required_run_keys, {"summary", "safe_outcome"}):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer)
|
||||
verdict = run.get("execution_verdict")
|
||||
mode = run.get("evidence_mode")
|
||||
accepted = run.get("accepted")
|
||||
source_revision = run.get("source_revision")
|
||||
environment_class = run.get("environment_class")
|
||||
command_id = run.get("command_id")
|
||||
source_hash = run.get("source_report_sha256")
|
||||
run_id = run.get("id")
|
||||
if not nonempty_text(run_id, 128) or run_id in seen_run_ids:
|
||||
add(diagnostics, "DUPLICATE_RUN_ID", f"{pointer}/id")
|
||||
elif isinstance(run_id, str):
|
||||
seen_run_ids.add(run_id)
|
||||
if command_id not in COMMAND_IDS:
|
||||
add(diagnostics, "UNKNOWN_COMMAND", f"{pointer}/command_id")
|
||||
if not isinstance(source_revision, str) or not re.fullmatch(r"[0-9a-f]{40}", source_revision):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/source_revision")
|
||||
elif source_revision != root_revision:
|
||||
add(diagnostics, "PROVENANCE_MISMATCH", f"{pointer}/source_revision")
|
||||
if not isinstance(environment_class, str) or not re.fullmatch(r"[a-z0-9-]{1,64}", environment_class):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/environment_class")
|
||||
elif environment_class != root_environment:
|
||||
add(diagnostics, "PROVENANCE_MISMATCH", f"{pointer}/environment_class")
|
||||
if not isinstance(source_hash, str) or not HEX_RE.fullmatch(source_hash):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"{pointer}/source_report_sha256")
|
||||
elif source_hash in seen_hashes:
|
||||
add(diagnostics, "DUPLICATE_REPORT_HASH", f"{pointer}/source_report_sha256")
|
||||
else:
|
||||
seen_hashes.add(source_hash)
|
||||
if verdict not in EXECUTION_VERDICTS or mode not in EVIDENCE_MODES:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", pointer)
|
||||
expected_accepted = verdict == "pass" and mode == "automated" and run.get("collector") == "capability-baseline-collector-v1"
|
||||
if type(accepted) is not bool or accepted is not expected_accepted:
|
||||
add(diagnostics, "NON_PASS_RECORDED_AS_PASS", pointer)
|
||||
flow_ids = run.get("flow_ids")
|
||||
if not isinstance(flow_ids, list) or not flow_ids or any(not isinstance(flow_id, str) for flow_id in flow_ids) or len(flow_ids) != len(set(flow_ids)):
|
||||
add(diagnostics, "MISSING_FLOW_EVIDENCE", f"{pointer}/flow_ids")
|
||||
continue
|
||||
for flow_id in flow_ids:
|
||||
if flow_id not in statuses:
|
||||
add(diagnostics, "UNKNOWN_FLOW_ID", f"{pointer}/flow_ids")
|
||||
elif accepted is True and statuses[flow_id] != "implemented":
|
||||
add(diagnostics, "NON_PASS_RECORDED_AS_PASS", f"{pointer}/flow_ids")
|
||||
elif accepted is True:
|
||||
evidenced.add(flow_id)
|
||||
for flow_id, status in statuses.items():
|
||||
if status == "implemented" and flow_id not in evidenced:
|
||||
add(diagnostics, "MISSING_FLOW_EVIDENCE", f"/inventory/{flow_id}")
|
||||
manual_results = results.get("manual_results")
|
||||
if not isinstance(manual_results, list) or len(manual_results) > 1000:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/manual_results")
|
||||
manual_results = []
|
||||
seen_checks: set[str] = set()
|
||||
for index, manual in enumerate(manual_results):
|
||||
pointer = f"/results/manual_results/{index}"
|
||||
required_manual_keys = {"check_id", "evidence_mode", "execution_verdict", "flow_ids", "next_evidence"}
|
||||
if not exact_keys(manual, required_manual_keys):
|
||||
add(diagnostics, "INVALID_MANUAL_EVIDENCE", pointer)
|
||||
continue
|
||||
check_id = manual.get("check_id")
|
||||
flow_ids = manual.get("flow_ids")
|
||||
if check_id in seen_checks or check_id not in checklist_checks:
|
||||
add(diagnostics, "INVALID_MANUAL_EVIDENCE", f"{pointer}/check_id")
|
||||
elif isinstance(check_id, str):
|
||||
seen_checks.add(check_id)
|
||||
if manual.get("evidence_mode") != "manual_only" or manual.get("execution_verdict") not in EXECUTION_VERDICTS:
|
||||
add(diagnostics, "INVALID_MANUAL_EVIDENCE", pointer)
|
||||
if not nonempty_text(manual.get("next_evidence"), 2048):
|
||||
add(diagnostics, "INVALID_MANUAL_EVIDENCE", f"{pointer}/next_evidence")
|
||||
if not isinstance(flow_ids, list) or not flow_ids or any(not isinstance(flow_id, str) or flow_id not in statuses for flow_id in flow_ids):
|
||||
add(diagnostics, "INVALID_MANUAL_EVIDENCE", f"{pointer}/flow_ids")
|
||||
elif isinstance(check_id, str) and check_id in checklist_checks:
|
||||
if checklist_checks[check_id]["flow_id"] not in flow_ids or checklist_checks[check_id]["verdict"] != manual.get("execution_verdict"):
|
||||
add(diagnostics, "INVALID_MANUAL_EVIDENCE", pointer)
|
||||
if seen_checks != set(checklist_checks):
|
||||
add(diagnostics, "INVALID_MANUAL_EVIDENCE", "/results/manual_results")
|
||||
defects = results.get("defects")
|
||||
if not isinstance(defects, list):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", "/results/defects")
|
||||
defects = []
|
||||
for index, defect in enumerate(defects):
|
||||
if not isinstance(defect, dict):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}")
|
||||
continue
|
||||
required_fields = ("id", "severity", "contract", "owner", "steps", "next_action", "flow_ids")
|
||||
if set(defect) != set(required_fields):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}")
|
||||
continue
|
||||
if defect.get("severity") not in {"Critical", "High", "Medium", "Low"}:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}/severity")
|
||||
if not isinstance(defect.get("steps"), list) or not 1 <= len(defect["steps"]) <= 20:
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}/steps")
|
||||
defect_flow_ids = defect.get("flow_ids")
|
||||
if not isinstance(defect_flow_ids, list) or not defect_flow_ids or any(not isinstance(flow_id, str) for flow_id in defect_flow_ids):
|
||||
add(diagnostics, "INVALID_SCHEMA_CONTRACT", f"/results/defects/{index}/flow_ids")
|
||||
defect_flow_ids = []
|
||||
for flow_id in defect_flow_ids:
|
||||
if flow_id not in statuses:
|
||||
add(diagnostics, "UNKNOWN_FLOW_ID", f"/results/defects/{index}/flow_ids")
|
||||
elif defect.get("severity") in SEVERE and statuses[flow_id] != "blocked":
|
||||
add(diagnostics, "SEVERE_DEFECT_FLOW_NOT_BLOCKED", f"/results/defects/{index}")
|
||||
|
||||
|
||||
def render(diagnostics: list[Diagnostic]) -> str:
|
||||
lines = [f"{item.code} pointer={item.pointer}" for item in sorted(set(diagnostics))]
|
||||
encoded = "\n".join(lines) + ("\n" if lines else "")
|
||||
raw = encoded.encode("utf-8")
|
||||
if len(raw) <= MAX_REPORT_BYTES:
|
||||
return encoded
|
||||
marker = "REPORT_TRUNCATED pointer=/\n"
|
||||
budget = MAX_REPORT_BYTES - len(marker.encode("utf-8"))
|
||||
return raw[:budget].decode("utf-8", "ignore") + marker
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--root", default=".")
|
||||
parser.add_argument("--manifest", required=True)
|
||||
parser.add_argument("--schema", required=True)
|
||||
args = parser.parse_args()
|
||||
root = Path(args.root)
|
||||
try:
|
||||
diagnostics, version, flow_count = validate(root, args.manifest, args.schema)
|
||||
except (OSError, ValueError, RecursionError):
|
||||
diagnostics, version, flow_count = [Diagnostic("VALIDATION_FAILED", "/")], None, 0
|
||||
if diagnostics:
|
||||
sys.stderr.write(render(diagnostics))
|
||||
return 1
|
||||
print(f"Capability baseline validation passed: baseline_version={version} flows={flow_count}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user