feat(import): resolve references and schema composition

This commit is contained in:
2026-08-29 08:54:26 +03:00
parent 6c2a3712d8
commit 55209a9bbc
46 changed files with 4848 additions and 437 deletions
+11 -8
View File
@@ -8,19 +8,22 @@ mod normalize_schema;
mod openapi3;
mod payload;
mod recommendations;
mod reference;
mod schema;
mod swagger2;
pub use model::{
ImportFinding, ImportFindingSeverity, ImportGroupPreview, ImportOperationCandidate,
ImportPreview, ImportSourcePreview, NORMALIZER_VERSION, NormalizationConfig, NormalizedFinding,
NormalizedIr, NormalizedOperation, NormalizedParameter, NormalizedReference, NormalizedSchema,
NormalizedSchemaKind, PROJECTION_VERSION, RestImportCandidate, RestImportDocument,
RestImportOperation, RestImportParameter, RestParameterLocation, SourceDigest, SourceIdentity,
SourceLocation, UnresolvedReference,
ExternalDocumentSnapshot, ImportFinding, ImportFindingSeverity, ImportGroupPreview,
ImportOperationCandidate, ImportPreview, ImportSourcePreview, NORMALIZER_VERSION,
NormalizationConfig, NormalizedFinding, NormalizedIr, NormalizedOperation, NormalizedParameter,
NormalizedReference, NormalizedSchema, NormalizedSchemaConstraints, NormalizedSchemaKind,
PROJECTION_VERSION, ResolvedReferenceEdge, ResolvedReferenceGraph, ResolvedReferenceNode,
RestImportCandidate, RestImportDocument, RestImportOperation, RestImportParameter,
RestParameterLocation, SourceDigest, SourceIdentity, SourceLocation, UnresolvedReference,
};
pub use normalize::{
ImportParseError, normalize_verified_document, preview_document, preview_document_legacy_v1,
preview_from_ir, validate_normalized_ir,
ImportParseError, external_reference_uris, normalize_verified_bundle,
normalize_verified_document, preview_document, preview_document_legacy_v1, preview_from_ir,
reference_uris, validate_normalized_ir,
};
pub use payload::operation_draft_from_candidate;
+82 -4
View File
@@ -9,10 +9,10 @@ use serde_json::Value;
/// The immutable contract used to normalize an OpenAPI source. These names
/// deliberately travel with an import job: changing either contract must not
/// silently reinterpret a pending preview.
pub const NORMALIZER_VERSION: &str = "normalized-ir-v2";
pub const PROJECTION_VERSION: &str = "preview-v2";
pub const NORMALIZER_VERSION: &str = "normalized-ir-v3";
pub const PROJECTION_VERSION: &str = "preview-v3";
#[derive(Clone, Debug, PartialEq, Eq, Serialize)]
#[derive(Clone, Debug, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize)]
pub struct SourceDigest(String);
impl SourceDigest {
@@ -97,6 +97,19 @@ pub struct NormalizationConfig {
pub max_collection_items: usize,
pub max_aliases: usize,
pub max_scalar_bytes: usize,
/// Maximum number of reference hops followed from one source location.
pub max_reference_depth: usize,
/// Maximum number of `$ref` occurrences inspected across the bundle.
pub max_references: usize,
/// Maximum number of immutable external documents supplied to the pure resolver.
pub max_reference_documents: usize,
/// Maximum number of nodes copied while expanding resolved references.
pub max_expanded_nodes: usize,
pub max_external_document_bytes: usize,
/// Signals that orchestration enabled external fetching. It never permits
/// I/O in this crate; it only distinguishes default-deny from a missing or
/// rejected supplied snapshot in exact findings.
pub external_references_enabled: bool,
}
impl Default for NormalizationConfig {
@@ -110,6 +123,12 @@ impl Default for NormalizationConfig {
max_collection_items: 10_000,
max_aliases: 128,
max_scalar_bytes: 256 * 1024,
max_reference_depth: 32,
max_references: 4_096,
max_reference_documents: 32,
max_expanded_nodes: 100_000,
max_external_document_bytes: 256 * 1024,
external_references_enabled: false,
}
}
}
@@ -139,6 +158,38 @@ pub struct UnresolvedReference {
pub location: SourceLocation,
}
/// Immutable external input for the pure reference resolver. The caller owns
/// URL policy and I/O; `crank-import` only consumes already verified bytes.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct ExternalDocumentSnapshot {
pub canonical_uri: String,
pub digest: SourceDigest,
pub document: String,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct ResolvedReferenceNode {
pub snapshot_digest: SourceDigest,
pub location: SourceLocation,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct ResolvedReferenceEdge {
pub source: ResolvedReferenceNode,
pub target: ResolvedReferenceNode,
pub recursive: bool,
}
#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)]
pub struct ResolvedReferenceGraph {
/// Sorted, deduplicated immutable dependency identities. Canonical URLs
/// deliberately do not enter the IR or public diagnostics.
#[serde(default)]
pub dependency_digests: Vec<SourceDigest>,
#[serde(default)]
pub edges: Vec<ResolvedReferenceEdge>,
}
/// An unresolved `$ref` preserved from the decoded source. This is deliberately
/// broader than schema references: path items, reusable parameters and other
/// object-level references remain available to a later resolution phase.
@@ -213,9 +264,34 @@ pub struct NormalizedSchema {
pub location: SourceLocation,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub description: Option<String>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub discriminator: Option<NormalizedDiscriminator>,
#[serde(default)]
pub constraints: NormalizedSchemaConstraints,
pub kind: NormalizedSchemaKind,
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
pub struct NormalizedSchemaConstraints {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub minimum: Option<f64>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub maximum: Option<f64>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub min_length: Option<u64>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub max_length: Option<u64>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub pattern: Option<String>,
}
#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)]
pub struct NormalizedDiscriminator {
pub property_name: String,
#[serde(default)]
pub mapping: BTreeMap<String, String>,
}
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case", tag = "type")]
pub enum NormalizedSchemaKind {
@@ -280,6 +356,8 @@ pub struct NormalizedIr {
pub paths: Vec<NormalizedPath>,
#[serde(default)]
pub unresolved_references: Vec<NormalizedReference>,
#[serde(default)]
pub reference_graph: ResolvedReferenceGraph,
pub source: ImportSourcePreview,
#[serde(default)]
pub operations: Vec<NormalizedOperation>,
@@ -416,7 +494,7 @@ fn empty_source_location() -> SourceLocation {
}
}
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)]
#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum RestParameterLocation {
Path,
+153 -15
View File
@@ -5,13 +5,13 @@ use thiserror::Error;
use crate::rest::{
model::{
CoverageDisposition, CoverageEntry, ImportFinding, ImportFindingSeverity,
ImportGroupPreview, ImportPreview, ImportSourcePreview, NORMALIZER_VERSION,
NormalizationConfig, NormalizedApiMetadata, NormalizedFinding, NormalizedIr,
NormalizedLiteral, NormalizedOperation, NormalizedParameter, NormalizedScalarKind,
NormalizedSchema, NormalizedSchemaKind, PROJECTION_VERSION, RestImportDocument,
RestImportOperation, RestImportParameter, SourceDigest, SourceIdentity, SourceLocation,
SourceSyntax,
CoverageDisposition, CoverageEntry, ExternalDocumentSnapshot, ImportFinding,
ImportFindingSeverity, ImportGroupPreview, ImportPreview, ImportSourcePreview,
NORMALIZER_VERSION, NormalizationConfig, NormalizedApiMetadata, NormalizedFinding,
NormalizedIr, NormalizedLiteral, NormalizedOperation, NormalizedParameter,
NormalizedScalarKind, NormalizedSchema, NormalizedSchemaKind, PROJECTION_VERSION,
RestImportDocument, RestImportOperation, RestImportParameter, SourceDigest, SourceIdentity,
SourceLocation, SourceSyntax,
},
openapi3,
payload::candidate_from_operation,
@@ -39,6 +39,25 @@ pub fn preview_document(document: &str) -> Result<ImportPreview, ImportParseErro
Ok(preview_from_ir(&ir))
}
/// Enumerates reference URIs without resolving or performing I/O. Intended
/// for an orchestration layer that materializes a bounded immutable bundle.
pub fn reference_uris(
document: &str,
config: &NormalizationConfig,
) -> Result<Vec<String>, ImportParseError> {
super::reference::reference_uris(document, config)
}
/// Enumerates references in an already materialized external document. This
/// keeps the historical primary-document API intact while applying the
/// external-document byte limit at an explicit call site.
pub fn external_reference_uris(
document: &str,
config: &NormalizationConfig,
) -> Result<Vec<String>, ImportParseError> {
super::reference::external_reference_uris(document, config)
}
/// Compatibility projection for jobs created before `NormalizedIr`. It is
/// deliberately isolated from the v2 pipeline and may be removed only after
/// the import-job TTL has elapsed.
@@ -83,6 +102,18 @@ pub fn normalize_verified_document(
document: &str,
digest: SourceDigest,
config: &NormalizationConfig,
) -> Result<NormalizedIr, ImportParseError> {
normalize_verified_bundle(document, digest, &[], config)
}
/// Normalizes a verified primary document together with immutable external
/// snapshots. This function is pure: callers must perform all URL policy,
/// fetching, artifact persistence and digest verification beforehand.
pub fn normalize_verified_bundle(
document: &str,
digest: SourceDigest,
snapshots: &[ExternalDocumentSnapshot],
config: &NormalizationConfig,
) -> Result<NormalizedIr, ImportParseError> {
config
.validate_versions()
@@ -99,6 +130,14 @@ pub fn normalize_verified_document(
};
let root = decode(document)?;
validate_value_limits(&root, config)?;
for snapshot in snapshots {
if snapshot.document.len() > config.max_external_document_bytes {
return Err(ImportParseError::LimitExceeded);
}
}
let resolution = super::reference::resolve(root, digest.clone(), snapshots, config)?;
let root = resolution.root;
validate_value_limits(&root, config)?;
let parsed = match root.get("openapi").and_then(Value::as_str) {
Some(version) if supported_oas_version(version) => openapi3::parse_document(&root)?,
@@ -112,7 +151,15 @@ pub fn normalize_verified_document(
if parsed.operations.is_empty() {
return Err(ImportParseError::NoMethods);
}
canonicalize(parsed, digest, config, root, source_syntax)
canonicalize(
parsed,
digest,
config,
root,
source_syntax,
resolution.graph,
resolution.findings,
)
}
pub fn preview_from_ir(ir: &NormalizedIr) -> ImportPreview {
@@ -202,6 +249,8 @@ fn canonicalize(
config: &NormalizationConfig,
root: Value,
source_syntax: SourceSyntax,
reference_graph: crate::rest::model::ResolvedReferenceGraph,
resolution_findings: Vec<NormalizedFinding>,
) -> Result<NormalizedIr, ImportParseError> {
let source = ImportSourcePreview {
format: document.format,
@@ -260,6 +309,7 @@ fn canonicalize(
}
})
.collect::<Vec<_>>();
findings.extend(resolution_findings);
let mut operations = document
.operations
.into_iter()
@@ -466,6 +516,40 @@ fn canonicalize(
right.location.pointer.as_str(),
))
});
let mut document_findings = Vec::with_capacity(findings.len());
for mut finding in findings.drain(..) {
let resolution_finding = matches!(
finding.code.as_str(),
"external_reference_disabled"
| "reference_target_missing"
| "reference_uri_malformed"
| "external_reference_unavailable"
| "reference_type_mismatch"
| "reference_graph_limit"
| "all_of_conflict"
| "unsupported_composition"
| "unsupported_discriminator"
);
let target = resolution_finding
.then(|| {
operations.iter_mut().find(|operation| {
!finding.location.pointer.is_empty()
&& (finding.location.pointer == operation.location.pointer
|| finding
.location
.pointer
.starts_with(&format!("{}/", operation.location.pointer)))
})
})
.flatten();
if let Some(operation) = target {
finding.operation_key = Some(operation.key.clone());
operation.findings.push(finding);
} else {
document_findings.push(finding);
}
}
findings = document_findings;
for operation in &mut operations {
sort_normalized_findings(&mut operation.findings);
}
@@ -501,10 +585,9 @@ fn canonicalize(
}
sort_normalized_findings(&mut operation.findings);
}
// Resolution is deliberately deferred. References already represented by
// an operation schema use that operation's blocker; all other references
// need a document-level blocker so omitted object-level semantics can
// never reach Apply silently.
// References left in the finite projection are precisely failures which
// the bounded resolver could not safely expand. Keep a blocker at the
// closest affected operation; unrelated operations remain actionable.
findings.extend(
unresolved_references
.iter()
@@ -529,7 +612,11 @@ fn canonicalize(
}),
);
sort_normalized_findings(&mut findings);
for finding in &findings {
for finding in findings.iter().chain(
operations
.iter()
.flat_map(|operation| operation.findings.iter()),
) {
if !coverage.iter().any(|entry| {
entry.construct_id == finding.construct_id
&& entry.location == finding.location
@@ -580,6 +667,7 @@ fn canonicalize(
.unwrap_or_default(),
paths: normalize_coverage::paths_from_operations(&operations, version),
unresolved_references,
reference_graph,
source,
operations,
findings,
@@ -593,9 +681,56 @@ pub fn validate_normalized_ir(ir: &NormalizedIr) -> Result<(), ImportParseError>
if ir.normalizer_version != NORMALIZER_VERSION || ir.projection_version != PROJECTION_VERSION {
return Err(ImportParseError::InvalidDocument);
}
validate_reference_graph(ir)?;
normalize_coverage::validate_coverage(ir)
}
fn validate_reference_graph(ir: &NormalizedIr) -> Result<(), ImportParseError> {
if ir
.reference_graph
.dependency_digests
.windows(2)
.any(|pair| pair[0] >= pair[1])
{
return Err(ImportParseError::InvalidDocument);
}
let allowed = ir
.reference_graph
.dependency_digests
.iter()
.chain(std::iter::once(&ir.source_identity.digest))
.map(SourceDigest::as_str)
.collect::<BTreeSet<_>>();
let edges = &ir.reference_graph.edges;
if edges
.windows(2)
.any(|pair| reference_edge_key(&pair[0]) >= reference_edge_key(&pair[1]))
|| edges.iter().any(|edge| {
!allowed.contains(edge.source.snapshot_digest.as_str())
|| !allowed.contains(edge.target.snapshot_digest.as_str())
|| !(edge.source.location.pointer.is_empty()
|| edge.source.location.pointer.starts_with('/'))
|| !(edge.target.location.pointer.is_empty()
|| edge.target.location.pointer.starts_with('/'))
})
{
return Err(ImportParseError::InvalidDocument);
}
Ok(())
}
fn reference_edge_key(
edge: &crate::rest::model::ResolvedReferenceEdge,
) -> (&str, &str, &str, &str, bool) {
(
edge.source.snapshot_digest.as_str(),
edge.source.location.pointer.as_str(),
edge.target.snapshot_digest.as_str(),
edge.target.location.pointer.as_str(),
edge.recursive,
)
}
fn coverage_disposition_rank(disposition: &CoverageDisposition) -> u8 {
match disposition {
CoverageDisposition::Mapped => 0,
@@ -650,7 +785,10 @@ fn legacy_schema_value(schema: &NormalizedSchema) -> Value {
let mut value = match &schema.kind {
NormalizedSchemaKind::Reference { reference } => serde_json::json!({"$ref": reference.uri}),
NormalizedSchemaKind::Composition { operator, variants } => {
serde_json::json!({operator: variants.iter().map(legacy_schema_value).collect::<Vec<_>>() })
serde_json::json!({
operator: variants.iter().map(legacy_schema_value).collect::<Vec<_>>(),
"x-crank-lossless-composition": true,
})
}
NormalizedSchemaKind::Object {
properties,
@@ -718,7 +856,7 @@ fn schema_contains_reference(schema: &NormalizedSchema, location: &SourceLocatio
}
}
fn validate_value_limits(
pub(super) fn validate_value_limits(
value: &Value,
config: &NormalizationConfig,
) -> Result<(), ImportParseError> {
@@ -1,8 +1,9 @@
use serde_json::Value;
use crate::rest::model::{
CoverageDisposition, CoverageEntry, NormalizedLiteral, NormalizedScalarKind, NormalizedSchema,
NormalizedSchemaKind, SourceLocation, UnresolvedReference,
CoverageDisposition, CoverageEntry, NormalizedDiscriminator, NormalizedLiteral,
NormalizedScalarKind, NormalizedSchema, NormalizedSchemaConstraints, NormalizedSchemaKind,
SourceLocation, UnresolvedReference,
};
use super::normalize_coverage::escape_pointer;
@@ -150,6 +151,35 @@ pub(super) fn typed_schema(
.get("description")
.and_then(Value::as_str)
.map(ToOwned::to_owned),
discriminator: value.get("discriminator").and_then(|value| {
let property_name = value.get("propertyName")?.as_str()?.to_owned();
let mapping = value
.get("mapping")
.and_then(Value::as_object)
.map(|mapping| {
mapping
.iter()
.filter_map(|(key, value)| {
value.as_str().map(|value| (key.clone(), value.to_owned()))
})
.collect()
})
.unwrap_or_default();
Some(NormalizedDiscriminator {
property_name,
mapping,
})
}),
constraints: NormalizedSchemaConstraints {
minimum: value.get("minimum").and_then(Value::as_f64),
maximum: value.get("maximum").and_then(Value::as_f64),
min_length: value.get("minLength").and_then(Value::as_u64),
max_length: value.get("maxLength").and_then(Value::as_u64),
pattern: value
.get("pattern")
.and_then(Value::as_str)
.map(ToOwned::to_owned),
},
kind,
}
}
+8
View File
@@ -133,6 +133,7 @@ fn parse_document_v2(root: &Value) -> Result<RestImportDocument, ImportParseErro
parsed_parameters.findings,
);
operation_parameters.extend(parsed_parameters.parameters);
deduplicate_parameters(&mut operation_parameters);
let operation_servers = parse_servers(
operation_value.get("servers"),
&format!("{operation_pointer}/servers"),
@@ -203,6 +204,13 @@ fn parse_document_v2(root: &Value) -> Result<RestImportDocument, ImportParseErro
})
}
fn deduplicate_parameters(parameters: &mut Vec<RestImportParameter>) {
let mut seen = std::collections::BTreeSet::new();
parameters.reverse();
parameters.retain(|parameter| seen.insert((parameter.name.clone(), parameter.location)));
parameters.reverse();
}
pub fn parse_document_legacy_v1(root: &Value) -> Result<RestImportDocument, ImportParseError> {
let version = root
.get("openapi")
+872
View File
@@ -0,0 +1,872 @@
use std::collections::{BTreeMap, BTreeSet};
use serde_json::{Map, Value};
use crate::rest::model::{
ExternalDocumentSnapshot, ImportFindingSeverity, NormalizationConfig, NormalizedFinding,
ResolvedReferenceEdge, ResolvedReferenceGraph, ResolvedReferenceNode, SourceDigest,
SourceLocation,
};
use super::{ImportParseError, normalize_coverage::escape_pointer};
pub(super) struct ResolutionResult {
pub root: Value,
pub graph: ResolvedReferenceGraph,
pub findings: Vec<NormalizedFinding>,
}
pub(super) fn reference_uris(
document: &str,
config: &NormalizationConfig,
) -> Result<Vec<String>, ImportParseError> {
reference_uris_with_max_bytes(document, config, config.max_bytes)
}
pub(super) fn external_reference_uris(
document: &str,
config: &NormalizationConfig,
) -> Result<Vec<String>, ImportParseError> {
reference_uris_with_max_bytes(document, config, config.max_external_document_bytes)
}
fn reference_uris_with_max_bytes(
document: &str,
config: &NormalizationConfig,
max_bytes: usize,
) -> Result<Vec<String>, ImportParseError> {
fn collect(value: &Value, uris: &mut BTreeSet<String>) {
match value {
Value::Object(object) => {
if let Some(reference) = object.get("$ref").and_then(Value::as_str) {
uris.insert(reference.to_owned());
}
for (key, child) in object {
if is_literal_payload_key(key) {
continue;
}
collect(child, uris);
}
}
Value::Array(items) => {
for child in items {
collect(child, uris);
}
}
_ => {}
}
}
if document.len() > max_bytes
|| super::normalize_limits::alias_count(document) > config.max_aliases
{
return Err(ImportParseError::LimitExceeded);
}
let root = decode_snapshot(document)?;
super::normalize::validate_value_limits(&root, config)?;
let mut uris = BTreeSet::new();
collect(&root, &mut uris);
Ok(uris.into_iter().collect())
}
struct Document {
uri: Option<String>,
digest: SourceDigest,
root: Value,
oas31: bool,
}
#[derive(Clone, Debug, Eq, Ord, PartialEq, PartialOrd)]
struct NodeKey {
digest: String,
pointer: String,
}
#[derive(Clone, Copy)]
struct TraversalLocation<'a> {
projection: &'a str,
origin: &'a str,
}
struct Resolver<'a> {
config: &'a NormalizationConfig,
documents: Vec<Document>,
by_uri: BTreeMap<String, usize>,
graph: ResolvedReferenceGraph,
findings: Vec<NormalizedFinding>,
references: usize,
expanded_nodes: usize,
}
pub(super) fn resolve(
root: Value,
primary_digest: SourceDigest,
snapshots: &[ExternalDocumentSnapshot],
config: &NormalizationConfig,
) -> Result<ResolutionResult, ImportParseError> {
if snapshots.len() > config.max_reference_documents {
return Err(ImportParseError::LimitExceeded);
}
let mut documents = vec![Document {
uri: None,
digest: primary_digest,
oas31: is_oas31(&root),
root,
}];
let mut by_uri = BTreeMap::new();
let mut dependency_digests = BTreeSet::new();
let primary_oas31 = documents[0].oas31;
for snapshot in snapshots {
if snapshot.canonical_uri.is_empty()
|| by_uri
.insert(snapshot.canonical_uri.clone(), documents.len())
.is_some()
{
return Err(ImportParseError::InvalidDocument);
}
if super::normalize_limits::alias_count(&snapshot.document) > config.max_aliases {
return Err(ImportParseError::LimitExceeded);
}
let root = decode_snapshot(&snapshot.document)?;
super::normalize::validate_value_limits(&root, config)?;
dependency_digests.insert(snapshot.digest.clone());
documents.push(Document {
uri: Some(snapshot.canonical_uri.clone()),
digest: snapshot.digest.clone(),
// External documents are fragments of the primary contract. In a
// 3.1 bundle they therefore use 3.1 `$ref` sibling semantics even
// when the fragment itself omits an `openapi` declaration.
oas31: primary_oas31 || is_oas31(&root),
root,
});
}
let mut resolver = Resolver {
config,
documents,
by_uri,
graph: ResolvedReferenceGraph {
dependency_digests: dependency_digests.into_iter().collect(),
edges: Vec::new(),
},
findings: Vec::new(),
references: 0,
expanded_nodes: 0,
};
let root = resolver.documents[0].root.clone();
let root = resolver.expand_value(
0,
root,
TraversalLocation {
projection: "",
origin: "",
},
0,
&mut Vec::new(),
)?;
resolver
.graph
.edges
.sort_by(|left, right| edge_key(left).cmp(&edge_key(right)));
resolver.graph.edges.dedup();
resolver.findings.sort_by(|left, right| {
(
&left.operation_key,
&left.construct_id,
&left.code,
&left.location.pointer,
)
.cmp(&(
&right.operation_key,
&right.construct_id,
&right.code,
&right.location.pointer,
))
});
resolver.findings.dedup();
Ok(ResolutionResult {
root,
graph: resolver.graph,
findings: resolver.findings,
})
}
impl Resolver<'_> {
fn expand_value(
&mut self,
document_index: usize,
value: Value,
location: TraversalLocation<'_>,
depth: usize,
stack: &mut Vec<NodeKey>,
) -> Result<Value, ImportParseError> {
if depth > 0 {
self.expanded_nodes = self.expanded_nodes.saturating_add(1);
if self.expanded_nodes > self.config.max_expanded_nodes {
self.push_finding("reference_graph_limit", location.projection);
return Ok(Value::Object(Map::new()));
}
}
match value {
Value::Object(mut object) => {
if let Some(reference) = object
.get("$ref")
.and_then(Value::as_str)
.map(str::to_owned)
{
return self.expand_reference(
document_index,
object,
&reference,
location,
depth,
stack,
);
}
let keys = object.keys().cloned().collect::<Vec<_>>();
for key in keys {
if let Some(child) = object.remove(&key) {
if is_literal_payload_key(&key) {
object.insert(key, child);
continue;
}
let child_pointer =
format!("{}/{}", location.projection, escape_pointer(&key));
let child_origin_pointer =
format!("{}/{}", location.origin, escape_pointer(&key));
object.insert(
key,
self.expand_value(
document_index,
child,
TraversalLocation {
projection: &child_pointer,
origin: &child_origin_pointer,
},
depth,
stack,
)?,
);
}
}
let composition_count = ["allOf", "oneOf", "anyOf"]
.into_iter()
.filter(|operator| object.contains_key(*operator))
.count();
if composition_count > 1 {
self.push_finding("unsupported_composition", location.projection);
}
if let Some(discriminator) = object.get("discriminator") {
let valid = discriminator
.get("propertyName")
.and_then(Value::as_str)
.is_some_and(|value| !value.is_empty())
&& discriminator.get("mapping").is_none_or(|mapping| {
mapping
.as_object()
.is_some_and(|mapping| mapping.values().all(Value::is_string))
})
&& (object.contains_key("oneOf") || object.contains_key("anyOf"));
if !valid {
self.push_finding("unsupported_discriminator", location.projection);
}
}
self.merge_all_of(object, location.projection)
}
Value::Array(items) => Ok(Value::Array(
items
.into_iter()
.enumerate()
.map(|(index, child)| {
self.expand_value(
document_index,
child,
TraversalLocation {
projection: &format!("{}/{index}", location.projection),
origin: &format!("{}/{index}", location.origin),
},
depth,
stack,
)
})
.collect::<Result<Vec<_>, _>>()?,
)),
other => Ok(other),
}
}
fn expand_reference(
&mut self,
document_index: usize,
mut source_object: Map<String, Value>,
reference: &str,
location: TraversalLocation<'_>,
depth: usize,
stack: &mut Vec<NodeKey>,
) -> Result<Value, ImportParseError> {
self.references = self.references.saturating_add(1);
if self.references > self.config.max_references || depth >= self.config.max_reference_depth
{
self.push_finding(
"reference_graph_limit",
&format!("{}/$ref", location.projection),
);
return Ok(Value::Object(source_object));
}
let target = self.target(document_index, reference);
if matches!(target, Err(TargetError::MalformedFragment)) {
self.push_finding(
"reference_uri_malformed",
&format!("{}/$ref", location.projection),
);
return Ok(Value::Object(source_object));
}
let Some((target_document, target_pointer)) = target.ok().flatten() else {
let external = reference.starts_with("http://") || reference.starts_with("https://");
let code = if external {
if self.config.external_references_enabled {
"external_reference_unavailable"
} else {
"external_reference_disabled"
}
} else {
"reference_target_missing"
};
self.push_finding(code, &format!("{}/$ref", location.projection));
return Ok(Value::Object(source_object));
};
let source = ResolvedReferenceNode {
snapshot_digest: self.documents[document_index].digest.clone(),
location: SourceLocation {
pointer: format!("{}/$ref", location.origin),
},
};
let target = ResolvedReferenceNode {
snapshot_digest: self.documents[target_document].digest.clone(),
location: SourceLocation {
pointer: target_pointer.clone(),
},
};
let key = NodeKey {
digest: target.snapshot_digest.as_str().to_owned(),
pointer: target_pointer.clone(),
};
let recursive = stack.contains(&key);
self.graph.edges.push(ResolvedReferenceEdge {
source,
target,
recursive,
});
if recursive {
// The edge is the lossless representation. Expansion stops here,
// producing an opaque object for the finite preview projection.
return Ok(Value::Object(Map::new()));
}
let Some(target_value) = self.documents[target_document]
.root
.pointer(&target_pointer)
.cloned()
else {
self.push_finding(
"reference_target_missing",
&format!("{}/$ref", location.projection),
);
return Ok(Value::Object(source_object));
};
if !target_value.is_object() {
self.push_finding(
"reference_type_mismatch",
&format!("{}/$ref", location.projection),
);
return Ok(Value::Object(source_object));
}
stack.push(key);
let mut expanded = self.expand_value(
target_document,
target_value,
TraversalLocation {
projection: location.projection,
origin: &target_pointer,
},
depth + 1,
stack,
)?;
stack.pop();
// OAS 3.1 Schema Objects permit siblings next to `$ref`; OAS 3.0 and
// Swagger Reference Objects ignore them. The source document version
// controls the semantics at the reference site.
source_object.remove("$ref");
if self.documents[document_index].oas31 && !source_object.is_empty() {
let Value::Object(expanded_object) = &mut expanded else {
self.push_finding(
"reference_type_mismatch",
&format!("{}/$ref", location.projection),
);
return Ok(Value::Object(source_object));
};
for (key, sibling) in source_object {
let child_pointer = format!("{}/{}", location.projection, escape_pointer(&key));
let child_origin_pointer = format!("{}/{}", location.origin, escape_pointer(&key));
expanded_object.insert(
key,
self.expand_value(
document_index,
sibling,
TraversalLocation {
projection: &child_pointer,
origin: &child_origin_pointer,
},
depth,
stack,
)?,
);
}
}
Ok(expanded)
}
fn target(
&self,
current: usize,
reference: &str,
) -> Result<Option<(usize, String)>, TargetError> {
let (document, fragment) = reference.split_once('#').unwrap_or((reference, ""));
let fragment = percent_decode(fragment)?;
let pointer = if fragment.is_empty() {
String::new()
} else if fragment.starts_with('/') {
fragment
} else {
return Ok(None);
};
if document.is_empty() {
return Ok(Some((current, pointer)));
}
let canonical = if has_uri_scheme(document) {
canonical_absolute_uri(document)
} else {
let Some(base) = self.documents[current].uri.as_deref() else {
return Ok(None);
};
join_relative(base, document)
};
Ok(canonical.and_then(|canonical| {
self.by_uri
.get(&canonical)
.copied()
.map(|index| (index, pointer))
}))
}
fn merge_all_of(
&mut self,
mut object: Map<String, Value>,
pointer: &str,
) -> Result<Value, ImportParseError> {
let branches = match object.remove("allOf") {
None => return Ok(Value::Object(object)),
Some(Value::Array(branches)) => branches,
Some(value) => {
object.insert("allOf".to_owned(), value);
self.push_finding("unsupported_composition", &format!("{pointer}/allOf"));
return Ok(Value::Object(object));
}
};
let original = branches.clone();
for branch in branches {
let Some(fields) = branch.as_object() else {
object.insert("allOf".to_owned(), Value::Array(original));
self.push_finding("all_of_conflict", &format!("{pointer}/allOf"));
return Ok(Value::Object(object));
};
if !merge_object(&mut object, fields) {
object.insert("allOf".to_owned(), Value::Array(original));
self.push_finding("all_of_conflict", &format!("{pointer}/allOf"));
return Ok(Value::Object(object));
}
}
Ok(Value::Object(object))
}
fn push_finding(&mut self, code: &str, pointer: &str) {
self.findings.push(NormalizedFinding {
code: code.to_owned(),
severity: ImportFindingSeverity::Error,
message: match code {
"external_reference_disabled" => "Внешняя ссылка отключена политикой импорта.",
"reference_target_missing" => {
"Цель ссылки отсутствует или имеет неверный JSON Pointer."
}
"external_reference_unavailable" => {
"Внешний snapshot недоступен или отклонён политикой импорта."
}
"reference_type_mismatch" => "Цель ссылки имеет неподдерживаемый тип.",
"reference_graph_limit" => "Граф ссылок превышает установленный предел.",
"all_of_conflict" => "Ветки allOf содержат несовместимые определения.",
"reference_uri_malformed" => "URI fragment ссылки содержит некорректное percent-кодирование.",
"unsupported_composition" => "Несколько операторов composition в одной schema не могут быть спроецированы без потерь.",
"unsupported_discriminator" => "Discriminator имеет неподдерживаемую или неполную структуру.",
_ => "Ссылка не может быть безопасно разрешена.",
}
.to_owned(),
construct_id: format!("source:{pointer}"),
location: SourceLocation {
pointer: pointer.to_owned(),
},
operation_key: None,
});
}
}
fn merge_object(target: &mut Map<String, Value>, source: &Map<String, Value>) -> bool {
for (key, value) in source {
match key.as_str() {
"properties" => {
let Some(source_properties) = value.as_object() else {
return false;
};
let properties = target
.entry(key.clone())
.or_insert_with(|| Value::Object(Map::new()));
let Some(target_properties) = properties.as_object_mut() else {
return false;
};
for (name, schema) in source_properties {
if let Some(existing) = target_properties.get(name) {
let (Some(existing), Some(schema)) =
(existing.as_object(), schema.as_object())
else {
return false;
};
let mut merged = existing.clone();
if !merge_object(&mut merged, schema) {
return false;
}
target_properties.insert(name.clone(), Value::Object(merged));
} else {
target_properties.insert(name.clone(), schema.clone());
}
}
}
"required" => {
let Some(source_required) = value.as_array() else {
return false;
};
let required = target
.entry(key.clone())
.or_insert_with(|| Value::Array(Vec::new()));
let Some(target_required) = required.as_array_mut() else {
return false;
};
target_required.extend(source_required.iter().cloned());
target_required.sort_by(|left, right| left.as_str().cmp(&right.as_str()));
target_required.dedup();
}
"minimum" | "maximum" => {
let Some(source_value) = value.as_f64() else {
return false;
};
if target.contains_key(key) && target.get(key).and_then(Value::as_f64).is_none() {
return false;
}
let merged = match (key.as_str(), target.get(key).and_then(Value::as_f64)) {
("minimum", Some(current)) => current.max(source_value),
("maximum", Some(current)) => current.min(source_value),
_ => source_value,
};
let Some(number) = serde_json::Number::from_f64(merged) else {
return false;
};
target.insert(key.clone(), Value::Number(number));
if constraint_contradiction(target, "minimum", "maximum") {
return false;
}
}
"minLength" | "maxLength" => {
let Some(source_value) = value.as_u64() else {
return false;
};
if target.contains_key(key) && target.get(key).and_then(Value::as_u64).is_none() {
return false;
}
let merged = match (key.as_str(), target.get(key).and_then(Value::as_u64)) {
("minLength", Some(current)) => current.max(source_value),
("maxLength", Some(current)) => current.min(source_value),
_ => source_value,
};
target.insert(key.clone(), Value::Number(merged.into()));
if integer_constraint_contradiction(target, "minLength", "maxLength") {
return false;
}
}
_ => {
if target.get(key).is_some_and(|existing| existing != value) {
return false;
}
target.insert(key.clone(), value.clone());
}
}
}
true
}
fn constraint_contradiction(target: &Map<String, Value>, minimum: &str, maximum: &str) -> bool {
match (
target.get(minimum).and_then(Value::as_f64),
target.get(maximum).and_then(Value::as_f64),
) {
(Some(minimum), Some(maximum)) => minimum > maximum,
_ => false,
}
}
fn integer_constraint_contradiction(
target: &Map<String, Value>,
minimum: &str,
maximum: &str,
) -> bool {
match (
target.get(minimum).and_then(Value::as_u64),
target.get(maximum).and_then(Value::as_u64),
) {
(Some(minimum), Some(maximum)) => minimum > maximum,
_ => false,
}
}
fn edge_key(edge: &ResolvedReferenceEdge) -> (&str, &str, &str, &str, bool) {
(
edge.source.snapshot_digest.as_str(),
&edge.source.location.pointer,
edge.target.snapshot_digest.as_str(),
&edge.target.location.pointer,
edge.recursive,
)
}
fn decode_snapshot(document: &str) -> Result<Value, ImportParseError> {
if let Ok(value) = serde_json::from_str(document) {
return Ok(value);
}
let yaml: serde_yaml::Value =
serde_yaml::from_str(document).map_err(|_| ImportParseError::InvalidDocument)?;
serde_json::to_value(yaml).map_err(|_| ImportParseError::InvalidDocument)
}
fn is_oas31(root: &Value) -> bool {
root.get("openapi")
.and_then(Value::as_str)
.is_some_and(|version| version.starts_with("3.1."))
}
fn join_relative(base: &str, relative: &str) -> Option<String> {
let base = UriReference::parse(base)?;
let relative = UriReference::parse(relative)?;
let scheme = relative.scheme.or(base.scheme)?;
let authority = if relative.scheme.is_some() || relative.authority.is_some() {
relative.authority
} else {
base.authority
};
let (path, query) = if relative.scheme.is_some() || relative.authority.is_some() {
(remove_dot_segments(&relative.path), relative.query)
} else if relative.path.is_empty() {
(base.path, relative.query.or(base.query))
} else if relative.path.starts_with('/') {
(remove_dot_segments(&relative.path), relative.query)
} else {
(
remove_dot_segments(&merge_paths(
&base.path,
authority.is_some(),
&relative.path,
)),
relative.query,
)
};
UriReference {
scheme: Some(scheme),
authority,
path,
query,
}
.render()
}
fn canonical_absolute_uri(uri: &str) -> Option<String> {
let reference = UriReference::parse(uri)?;
let scheme = reference.scheme?;
UriReference {
scheme: Some(scheme),
authority: reference.authority,
path: remove_dot_segments(&reference.path),
query: reference.query,
}
.render()
}
#[derive(Clone, Debug)]
struct UriReference<'a> {
scheme: Option<&'a str>,
authority: Option<&'a str>,
path: String,
query: Option<&'a str>,
}
impl<'a> UriReference<'a> {
fn parse(value: &'a str) -> Option<Self> {
if value.contains('\\') || value.contains('#') {
return None;
}
let (without_query, query) = value
.split_once('?')
.map_or((value, None), |(path, query)| (path, Some(query)));
let (scheme, rest) = if let Some(index) = without_query.find(':') {
let candidate = &without_query[..index];
if is_uri_scheme(candidate) {
(Some(candidate), &without_query[index + 1..])
} else {
(None, without_query)
}
} else {
(None, without_query)
};
let (authority, path) = if let Some(rest) = rest.strip_prefix("//") {
match rest.find('/') {
Some(index) => (Some(&rest[..index]), rest[index..].to_owned()),
None => (Some(rest), String::new()),
}
} else {
(None, rest.to_owned())
};
Some(Self {
scheme,
authority,
path,
query,
})
}
fn render(&self) -> Option<String> {
let scheme = self.scheme?;
let scheme = scheme.to_ascii_lowercase();
let mut value = format!("{scheme}:");
if let Some(authority) = self.authority {
value.push_str("//");
value.push_str(&canonical_authority(authority, &scheme));
}
if self.authority.is_some() && self.path.is_empty() {
value.push('/');
} else {
value.push_str(&self.path);
}
if let Some(query) = self.query {
value.push('?');
value.push_str(query);
}
Some(value)
}
}
fn canonical_authority(authority: &str, scheme: &str) -> String {
let authority = authority.to_ascii_lowercase();
let default_port = match scheme {
"http" => Some("80"),
"https" => Some("443"),
_ => None,
};
if let Some(default_port) = default_port
&& let Some((host, port)) = authority.rsplit_once(':')
&& port == default_port
&& (!host.contains(':') || host.ends_with(']'))
{
return host.to_owned();
}
authority
}
fn has_uri_scheme(value: &str) -> bool {
value
.split_once(':')
.is_some_and(|(candidate, _)| is_uri_scheme(candidate))
}
fn is_uri_scheme(candidate: &str) -> bool {
let Some(first) = candidate.as_bytes().first() else {
return false;
};
first.is_ascii_alphabetic()
&& candidate
.bytes()
.all(|byte| byte.is_ascii_alphanumeric() || matches!(byte, b'+' | b'-' | b'.'))
}
fn merge_paths(base_path: &str, has_authority: bool, relative_path: &str) -> String {
match base_path.rfind('/') {
Some(index) => format!("{}{}", &base_path[..=index], relative_path),
None if has_authority => format!("/{relative_path}"),
None => relative_path.to_owned(),
}
}
fn remove_dot_segments(path: &str) -> String {
let leading_slash = path.starts_with('/');
let trailing_slash = path.ends_with('/');
let mut output = Vec::new();
for segment in path.split('/') {
match segment {
"." => {}
".." => {
output.pop();
}
_ => output.push(segment),
}
}
let mut result = output.join("/");
if leading_slash && !result.starts_with('/') {
result.insert(0, '/');
}
if trailing_slash && !result.ends_with('/') {
result.push('/');
}
result
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
enum TargetError {
MalformedFragment,
}
fn percent_decode(fragment: &str) -> Result<String, TargetError> {
let bytes = fragment.as_bytes();
let mut decoded = Vec::with_capacity(bytes.len());
let mut index = 0;
while index < bytes.len() {
if bytes[index] != b'%' {
decoded.push(bytes[index]);
index += 1;
continue;
}
let Some(high) = bytes.get(index + 1).and_then(|byte| hex_value(*byte)) else {
return Err(TargetError::MalformedFragment);
};
let Some(low) = bytes.get(index + 2).and_then(|byte| hex_value(*byte)) else {
return Err(TargetError::MalformedFragment);
};
decoded.push((high << 4) | low);
index += 3;
}
String::from_utf8(decoded).map_err(|_| TargetError::MalformedFragment)
}
fn hex_value(byte: u8) -> Option<u8> {
match byte {
b'0'..=b'9' => Some(byte - b'0'),
b'a'..=b'f' => Some(byte - b'a' + 10),
b'A'..=b'F' => Some(byte - b'A' + 10),
_ => None,
}
}
fn is_literal_payload_key(key: &str) -> bool {
matches!(key, "example" | "examples" | "default" | "enum" | "const") || key.starts_with("x-")
}
+27 -2
View File
@@ -32,8 +32,33 @@ pub fn schema_from_openapi(
let Some(value) = value else {
return primitive(SchemaKind::String, required, description);
};
// NormalizedIR keeps composition typed; the legacy preview adapter retains
// its historical first-branch projection for pending v1 jobs.
// Only v3 NormalizedIR emits the private marker. Pending legacy-v1 jobs
// retain their historical first-branch behavior, while modern oneOf/anyOf
// is projected losslessly through crank-schema's existing Oneof shape.
if value
.get("x-crank-lossless-composition")
.and_then(Value::as_bool)
== Some(true)
&& let Some(items) = value
.get("oneOf")
.or_else(|| value.get("anyOf"))
.and_then(Value::as_array)
{
return Schema {
kind: SchemaKind::Oneof,
description: description.or_else(|| text(value, "description")),
required,
nullable: nullable(value),
default_value: value.get("default").cloned(),
fields: BTreeMap::new(),
items: None,
enum_values: Vec::new(),
variants: items
.iter()
.map(|item| schema_from_openapi(Some(item), true, None))
.collect(),
};
}
let resolved = collapse_composition(value);
if let Some(values) = resolved.get("enum").and_then(Value::as_array) {
+8
View File
@@ -107,6 +107,7 @@ fn parse_document_v2(root: &Value) -> Result<RestImportDocument, ImportParseErro
parsed_parameters.findings,
);
operation_parameters.extend(parsed_parameters.parameters);
deduplicate_parameters(&mut operation_parameters);
let tags = tags(
operation_value.get("tags"),
&format!("{path_pointer}/{method_name}/tags"),
@@ -183,6 +184,13 @@ fn parse_document_v2(root: &Value) -> Result<RestImportDocument, ImportParseErro
})
}
fn deduplicate_parameters(parameters: &mut Vec<RestImportParameter>) {
let mut seen = std::collections::BTreeSet::new();
parameters.reverse();
parameters.retain(|parameter| seen.insert((parameter.name.clone(), parameter.location)));
parameters.reverse();
}
pub fn parse_document_legacy_v1(root: &Value) -> Result<RestImportDocument, ImportParseError> {
let title = root
.pointer("/info/title")