From 54fd112fde6f7e55e69ca75a4a46c06109c4fe4a Mon Sep 17 00:00:00 2001 From: "glm-5.2" Date: Sat, 15 Aug 2026 13:39:05 +0000 Subject: [PATCH] Remove v0.1.0 custom-keyword machinery (step 8) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The BAST parser (step 3) and BAST-native validator (step 5) replaced the v0.1.0 custom-keyword accessor layer; step 7 moved the builder to BAST output. This step removes the now-dead code: Removed from src/schema.rs: - get_alktype_kind / get_alktype_kind_enum / get_alktype_kind_loose / get_alktype_kind_loose_enum (replaced by the BAST parser's kind dispatch) - normalize_refs / inline_union_variant_refs + helpers (BAST refs are always #/$defs/; resolution is a single hash lookup, variant refs resolve lazily) - parse_encoding / parse_align / parse_max_length / parse_endian (bast.rs has its own BAST-property-form copies) - parse_discriminator + DiscriminatorKind (replaced by bast::BastDiscriminator; builder has its own Discriminator enum) - resolve_ref / resolve_ref_or_inline (replaced by BastDoc::lookup_def / resolve_typeref) - FromStr impl, as_str, Endian::from_schema, ALKTYPE_PREFIX, BYTE_DISCRIMINATOR_TYPES, and the associated unit tests Kept: AlkTypeKind enum + methods (type_size, natural_alignment, is_fixed_size, needs_endian, is_composite, is_variable_length, to_bast_str, from_bast_str), Display (now backed by to_bast_str), Endian, VariableEncoding, U32_SIZE, DISCRIMINATOR_PATH. src/lib.rs: dropped the 13 schema::* helper re-exports and DiscriminatorKind from the public surface; kept Endian, AlkTypeKind, VariableEncoding. Doc/comment updates: bast.rs, builder.rs, engine.rs, error.rs — removed references to the deleted functions and the AlkType:* keyword form. The jsonschema crate remains a dependency (validate_json path + BAST meta-schema validation); build_validator was already repurposed in step 6 (no custom keywords). Verification: - cargo test --release: 389 pass (312 lib + 77 integration) - cargo clippy --all-targets -- -D warnings: clean - cargo doc --no-deps: clean - cargo build --target wasm32-unknown-unknown --release: clean --- src/bast.rs | 37 +-- src/builder.rs | 8 +- src/engine.rs | 6 +- src/error.rs | 4 +- src/lib.rs | 6 +- src/schema.rs | 835 ++----------------------------------------------- 6 files changed, 51 insertions(+), 845 deletions(-) diff --git a/src/bast.rs b/src/bast.rs index 646897e..64fbad0 100644 --- a/src/bast.rs +++ b/src/bast.rs @@ -1,7 +1,6 @@ //! BAST document parser — the typed surface over a BAST (Binary Abstract //! Syntax Tree) document that the layout engines, materializer, and -//! BAST-native validator walk instead of the v0.1.0 `get_alktype_kind*` -//! custom-keyword accessors. +//! BAST-native validator walk. //! //! This is step 3 of the BAST pivot //! ([`docs/plans/bast-implementation.md`](../docs/plans/bast-implementation.md)). @@ -19,11 +18,10 @@ //! parse allocation-free beyond the small typed nodes themselves. //! //! `$ref` resolution is a single hash lookup against `$defs` — BAST refs -//! are always full JSON Pointers restricted to `#/$defs/` (no -//! `normalize_refs`, no `inline_union_variant_refs`; both are removed in -//! step 8). Union variant `$ref`s are resolved **lazily** by the -//! materializer/validator via [`BastDoc::resolve_typeref`]; the parser -//! only records the [`BastRef`] target name. +//! are always full JSON Pointers restricted to `#/$defs/`. Union +//! variant `$ref`s are resolved **lazily** by the materializer/validator +//! via [`BastDoc::resolve_typeref`]; the parser only records the +//! [`BastRef`] target name. //! //! ## Untrusted input //! @@ -91,8 +89,8 @@ impl<'a> BastDoc<'a> { /// Look up a raw `$defs` entry by name. Returns the raw JSON node. /// - /// The single hash lookup that replaces v0.1.0's `normalize_refs` + - /// `resolve_ref` machinery. BAST refs are always `#/$defs/`; + /// The single hash lookup that replaces v0.1.0's ref-normalization + /// machinery. BAST refs are always `#/$defs/`; /// no external refs, no fragment-only pointers, no bare names. pub fn lookup_def(&self, name: &str) -> Result<&'a Value, AlkTypeError> { Self::lookup_def_raw(self.root, name) @@ -102,8 +100,8 @@ impl<'a> BastDoc<'a> { /// /// Used lazily by the materializer and BAST-native validator (step 5) /// for union variant dispatch and `$ref` fields. The parser does not - /// inline refs at construction time — this is the replacement for - /// v0.1.0's `inline_union_variant_refs`. + /// inline refs at construction time — variant refs stay as + /// [`BastRef`]s and resolve on demand. pub fn resolve_ref(&self, r: &BastRef<'a>) -> Result, AlkTypeError> { let raw = self.lookup_def(r.name())?; BastDef::parse(raw, r.name(), "") @@ -440,8 +438,7 @@ impl<'a> BastField<'a> { /// discriminator. /// /// Variant `$ref`s are resolved **lazily** by the materializer/validator -/// via [`BastDoc::resolve_typeref`] — no compile-time inlining (the -/// replacement for v0.1.0's `inline_union_variant_refs`). +/// via [`BastDoc::resolve_typeref`] — no compile-time inlining. #[derive(Debug, Clone)] pub struct BastUnion<'a> { endian: Endian, @@ -774,8 +771,8 @@ impl<'a> BastType<'a> { /// A `$ref` to a named `$defs` entry, restricted to `#/$defs/`. /// /// The restriction keeps resolution a single hash lookup -/// ([`BastDoc::lookup_def`]) and eliminates the `normalize_refs` step -/// the v0.1.0 engine needed for TypeBox's bare-name refs. +/// ([`BastDoc::lookup_def`]); the v0.1.0 engine needed a normalization +/// pass for TypeBox's bare-name refs, but BAST forbids them. #[derive(Debug, Clone)] pub struct BastRef<'a> { name: &'a str, @@ -892,11 +889,11 @@ impl<'a> BastRecord<'a> { // Shared annotation parsers (BAST-property form). // // These read type-level/field-level properties (`endian`, `align`, -// `encoding`, `maxLength`) directly off a BAST node, instead of the -// v0.1.0 keyword-value objects that `schema::parse_*` read. Their -// *semantics* are unchanged (ADR-003); only the input location moved. -// They're `fn`s rather than methods on the typed nodes so the layout -// engines can call them on raw `&Value` during the step 4 migration. +// `encoding`, `maxLength`) directly off a BAST node. Their *semantics* +// are unchanged (ADR-003); only the input location moved from the +// v0.1.0 keyword-value objects to BAST type-level properties. They're +// `fn`s rather than methods on the typed nodes so the layout engines +// can call them on raw `&Value` during the step 4 migration. // --------------------------------------------------------------------------- fn parse_endian_opt(node: &Value) -> Option { diff --git a/src/builder.rs b/src/builder.rs index 0573e1e..d8b8831 100644 --- a/src/builder.rs +++ b/src/builder.rs @@ -10,9 +10,9 @@ //! Feed to [`crate::AlkTypeEngine::compile`]. //! - **Standard JSON Schema** (JSON validation) — `Schema::object().field(...)` //! produces `{ "type": "object", "properties": {...}, "required": [...] }`. -//! No BAST `kind`, no `AlkType:*` keywords. Feed to a standard -//! `jsonschema::Validator` (or [`crate::AlkTypeEngine::compile`] with a -//! JSON Schema for the `validate_json` path). +//! No BAST `kind`, no custom keywords — a plain JSON Schema. Feed to a +//! standard `jsonschema::Validator` (or [`crate::AlkTypeEngine::compile`] +//! with a JSON Schema for the `validate_json` path). //! //! The construction API distinguishes BAST kinds from standard JSON Schema //! types via naming conventions (`string()` is the BAST primitive, @@ -705,7 +705,7 @@ fn discriminator_value(disc: &Discriminator) -> Value { } /// TUnion discriminator, builder-friendly form. Mirrors -/// [`crate::schema::DiscriminatorKind`] but constructed via the builder +/// [`crate::bast::BastDiscriminator`] but constructed via the builder /// API. #[derive(Debug, Clone)] pub enum Discriminator { diff --git a/src/engine.rs b/src/engine.rs index 386c8f7..bc031ea 100644 --- a/src/engine.rs +++ b/src/engine.rs @@ -984,10 +984,8 @@ mod tests { // // Note: step 4 wires the layout layer to BAST. The `validate_bytes` // value-constraint enforcement (maxLength, enum bounds) is step 5's - // concern (the BAST-native validator). Until step 5 lands, the - // validator built from a BAST doc sees no `AlkType:*` custom keywords, - // so jsonschema performs only structural validation. These tests - // cover the materialization + structural path. + // concern (the BAST-native validator). These tests cover the + // materialization + structural path. fn chunk_header_doc() -> Value { json!({ diff --git a/src/error.rs b/src/error.rs index 9bc474e..fd80702 100644 --- a/src/error.rs +++ b/src/error.rs @@ -9,8 +9,8 @@ use std::fmt; /// Errors produced by the alktype engine across all phases. #[derive(Debug)] pub enum AlkTypeError { - /// Schema parsing errors — invalid JSON, missing required keywords, - /// unknown `AlkType:*` kinds, malformed annotations. + /// Schema parsing errors — invalid JSON, missing required properties, + /// unknown BAST kinds, malformed annotations. Schema(String), /// Offset computation errors — field not found, type not supported diff --git a/src/lib.rs b/src/lib.rs index 0358bbd..6c42ca3 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -63,11 +63,7 @@ pub use engine::{LayoutMode, AlkTypeEngine}; pub use error::AlkTypeError; pub use layout_builder::{FieldPosition, LayoutBuilder, PackedLayout}; pub use offset_map::{ByteRange, OffsetMap}; -pub use schema::{ - get_alktype_kind_loose, get_alktype_kind_loose_enum, inline_union_variant_refs, normalize_refs, - parse_align, parse_discriminator, parse_encoding, parse_endian, parse_max_length, resolve_ref, - resolve_ref_or_inline, DiscriminatorKind, Endian, AlkTypeKind, VariableEncoding, -}; +pub use schema::{Endian, AlkTypeKind, VariableEncoding}; pub use sequential_reader::{FieldValue, SequentialReader}; pub use tunion::UnionDispatch; pub use validation::build_validator; diff --git a/src/schema.rs b/src/schema.rs index 257e458..229083b 100644 --- a/src/schema.rs +++ b/src/schema.rs @@ -1,33 +1,26 @@ -//! Schema layer: AlkType kind detection, annotation parsing, `$ref` -//! normalization, and the `Endian` enum. +//! Schema layer: the `AlkTypeKind` enum, `Endian`/`VariableEncoding` +//! annotations, and shared constants. //! -//! Per ADR-097 and the schema-layer spec. This module provides the -//! foundational types and functions that every other module depends on: -//! `AlkType:*` kind detection, fixed byte-size lookups, natural alignment, -//! schema annotation parsing (`encoding`, `align`, `maxLength`, `endian`), -//! TUnion discriminator parsing, and `$ref` normalization from TypeBox -//! bare-name refs to JSON Pointer refs. +//! Under the BAST pivot the v0.1.0 custom-keyword accessors +//! (`get_alktype_kind*`, `normalize_refs`, `inline_union_variant_refs`, +//! `parse_*`, `resolve_ref*`, `DiscriminatorKind`) were removed — the +//! BAST parser ([`crate::bast`]) is the typed surface every engine +//! module walks. This module retains only the foundational types that +//! the layout/data-access/materialize/tunion/builder layers depend on: +//! the kind enum (with its BAST string mapping), the two annotation +//! enums, and the shared byte-layout constants. use crate::error::AlkTypeError; -use serde_json::Value; use std::fmt; -use std::str::FromStr; - -const ALKTYPE_PREFIX: &str = "AlkType:"; pub(crate) const U32_SIZE: usize = 4; pub(crate) const DISCRIMINATOR_PATH: &str = "__discriminator"; -const BYTE_DISCRIMINATOR_TYPES: &[AlkTypeKind] = &[ - AlkTypeKind::Uint8, - AlkTypeKind::Uint16, - AlkTypeKind::Uint32, -]; - -/// The 19 `AlkType:*` kinds recognized by the engine. +/// The 19 BAST kinds recognized by the engine. /// -/// Each variant corresponds to a `AlkType:` JSON Schema keyword. -/// The enum provides compile-time exhaustiveness checking and integer +/// Each variant corresponds to a lowercase BAST kind string +/// ([`AlkTypeKind::to_bast_str`], e.g. `"uint32"` ↔ `Uint32`). The enum +/// provides compile-time exhaustiveness checking and integer /// discriminant dispatch (jump table) instead of string comparison. #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] pub enum AlkTypeKind { @@ -53,31 +46,6 @@ pub enum AlkTypeKind { } impl AlkTypeKind { - /// The JSON Schema keyword string, e.g. `"AlkType:Int8"`. - pub fn as_str(self) -> &'static str { - match self { - AlkTypeKind::Int8 => "AlkType:Int8", - AlkTypeKind::Int16 => "AlkType:Int16", - AlkTypeKind::Int32 => "AlkType:Int32", - AlkTypeKind::Int64 => "AlkType:Int64", - AlkTypeKind::Uint8 => "AlkType:Uint8", - AlkTypeKind::Uint16 => "AlkType:Uint16", - AlkTypeKind::Uint32 => "AlkType:Uint32", - AlkTypeKind::Uint64 => "AlkType:Uint64", - AlkTypeKind::Float32 => "AlkType:Float32", - AlkTypeKind::Float64 => "AlkType:Float64", - AlkTypeKind::Boolean => "AlkType:Boolean", - AlkTypeKind::Enum => "AlkType:Enum", - AlkTypeKind::String => "AlkType:String", - AlkTypeKind::Bytes => "AlkType:Bytes", - AlkTypeKind::Struct => "AlkType:Struct", - AlkTypeKind::Union => "AlkType:Union", - AlkTypeKind::Array => "AlkType:Array", - AlkTypeKind::Record => "AlkType:Record", - AlkTypeKind::Timestamp => "AlkType:Timestamp", - } - } - /// Fixed byte size, or `None` for variable-size / composite kinds. pub fn type_size(self) -> Option { match self { @@ -177,9 +145,7 @@ impl AlkTypeKind { /// The lowercase BAST kind string, e.g. `"uint32"` (D-BAST-002). /// /// This is the string used in BAST documents (`{ "kind": "uint32" }`) - /// and the value the parser and validator dispatch on. Distinct from - /// [`as_str`](Self::as_str), which returns the v0.1.0 `AlkType:Uint32` - /// keyword form. + /// and the value the parser and validator dispatch on. pub fn to_bast_str(self) -> &'static str { match self { AlkTypeKind::Int8 => "int8", @@ -207,9 +173,7 @@ impl AlkTypeKind { /// Parse a lowercase BAST kind string into an `AlkTypeKind` /// (D-BAST-002). Returns `AlkTypeError::Schema` for unknown strings. /// - /// This is the inverse of [`to_bast_str`](Self::to_bast_str) and is - /// distinct from the v0.1.0 `FromStr` impl, which parses the - /// `AlkType:Uint32` keyword form. + /// This is the inverse of [`to_bast_str`](Self::to_bast_str). pub fn from_bast_str(s: &str) -> Result { match s { "int8" => Ok(AlkTypeKind::Int8), @@ -240,64 +204,10 @@ impl AlkTypeKind { impl fmt::Display for AlkTypeKind { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - f.write_str(self.as_str()) + f.write_str(self.to_bast_str()) } } -impl FromStr for AlkTypeKind { - type Err = AlkTypeError; - - fn from_str(s: &str) -> Result { - match s { - "AlkType:Int8" => Ok(AlkTypeKind::Int8), - "AlkType:Int16" => Ok(AlkTypeKind::Int16), - "AlkType:Int32" => Ok(AlkTypeKind::Int32), - "AlkType:Int64" => Ok(AlkTypeKind::Int64), - "AlkType:Uint8" => Ok(AlkTypeKind::Uint8), - "AlkType:Uint16" => Ok(AlkTypeKind::Uint16), - "AlkType:Uint32" => Ok(AlkTypeKind::Uint32), - "AlkType:Uint64" => Ok(AlkTypeKind::Uint64), - "AlkType:Float32" => Ok(AlkTypeKind::Float32), - "AlkType:Float64" => Ok(AlkTypeKind::Float64), - "AlkType:Boolean" => Ok(AlkTypeKind::Boolean), - "AlkType:Enum" => Ok(AlkTypeKind::Enum), - "AlkType:String" => Ok(AlkTypeKind::String), - "AlkType:Bytes" => Ok(AlkTypeKind::Bytes), - "AlkType:Struct" => Ok(AlkTypeKind::Struct), - "AlkType:Union" => Ok(AlkTypeKind::Union), - "AlkType:Array" => Ok(AlkTypeKind::Array), - "AlkType:Record" => Ok(AlkTypeKind::Record), - "AlkType:Timestamp" => Ok(AlkTypeKind::Timestamp), - other => Err(AlkTypeError::Schema(format!( - "unknown AlkType kind: {other}" - ))), - } - } -} - -/// Returns the `AlkType:*` kind string if the schema node declares one. -/// Returns `None` if the node has no `AlkType:*` keyword. -/// -/// A AlkType kind is recognized when the schema object has a key starting -/// with `AlkType:` whose value is `true`. (Object form with annotations -/// like `{ "encoding": "..." }` is handled by the annotation parsers, not -/// here — `get_alktype_kind` only checks for the presence of the keyword.) -pub fn get_alktype_kind(node: &Value) -> Option<&str> { - let obj = node.as_object()?; - for key in obj.keys() { - if key.starts_with(ALKTYPE_PREFIX) && obj.get(key) == Some(&Value::Bool(true)) { - return Some(key.as_str()); - } - } - None -} - -/// Returns the `AlkTypeKind` enum variant if the schema node declares one -/// (boolean form only, like `get_alktype_kind`). -pub fn get_alktype_kind_enum(node: &Value) -> Option { - get_alktype_kind(node).and_then(|s| s.parse().ok()) -} - /// Byte endianness for multi-byte integer and float fields. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Endian { @@ -305,14 +215,6 @@ pub enum Endian { Big, } -impl Endian { - /// Parse from the schema's `"endian"` annotation. Defaults to `Little` - /// if the annotation is absent or unrecognized. - pub fn from_schema(schema: &Value) -> Self { - parse_endian(schema) - } -} - /// The encoding strategy for a variable-length type. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum VariableEncoding { @@ -324,292 +226,9 @@ pub enum VariableEncoding { OffsetIndirect, } -/// Parse the `"encoding"` annotation from a variable-length type's keyword value. -/// -/// The keyword value may be `true` (shorthand for length-prefixed) or an -/// object with an `"encoding"` field. Defaults to `LengthPrefixed` when -/// absent or unrecognized. -pub fn parse_encoding(keyword_value: &Value) -> VariableEncoding { - match keyword_value { - Value::Bool(true) => VariableEncoding::LengthPrefixed, - Value::Object(obj) => { - let encoding = obj.get("encoding").and_then(Value::as_str); - match encoding { - Some("offset-indirect") => VariableEncoding::OffsetIndirect, - _ => VariableEncoding::LengthPrefixed, - } - } - _ => VariableEncoding::LengthPrefixed, - } -} - -/// Parse the `"align"` annotation from a schema node. Returns `None` if not -/// specified or not a non-negative integer. -pub fn parse_align(node: &Value) -> Option { - let n = node.as_object()?.get("align")?.as_u64()?; - Some(n as usize) -} - -/// Parse the `"maxLength"` annotation (standard JSON Schema keyword). -/// Returns `None` if not specified or not a non-negative integer. -pub fn parse_max_length(node: &Value) -> Option { - let n = node.as_object()?.get("maxLength")?.as_u64()?; - Some(n as usize) -} - -/// Parse the `"endian"` annotation. Defaults to `Little` if absent or -/// unrecognized. Operates on any node, not just the root. -pub fn parse_endian(node: &Value) -> Endian { - match node - .as_object() - .and_then(|o| o.get("endian")) - .and_then(Value::as_str) - { - Some("big") => Endian::Big, - _ => Endian::Little, - } -} - -/// The kind of TUnion discriminator. -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum DiscriminatorKind { - /// Byte-offset discriminator: a fixed-size integer at a known byte offset. - /// Mapping keys are stringified integers. Used by SFTP type bytes and - /// call protocol event types. - Byte { - /// Byte position of the discriminator within the union's buffer. - offset: usize, - /// The `AlkType:*` kind of the discriminator (typically - /// `AlkType:Uint8`). - disc_type: AlkTypeKind, - }, - /// Field-name discriminator: a named field within the struct. Mapping keys - /// are string values matching the discriminator field's value. The - /// typedef.ts pattern. - Field { - /// The field name that holds the discriminator value. - name: String, - }, -} - -/// Parse the `"discriminator"` annotation from a TUnion schema node. -/// -/// Returns [`AlkTypeError::Schema`] for malformed discriminators (unknown -/// `kind`, missing required `name`, or an unsupported discriminator `type`). -pub fn parse_discriminator(node: &Value) -> Result { - let obj = node.as_object().ok_or_else(|| { - AlkTypeError::Schema("discriminator requires a schema object".to_string()) - })?; - let disc = obj.get("discriminator").ok_or_else(|| { - AlkTypeError::Schema("union is missing 'discriminator' annotation".to_string()) - })?; - let disc_obj = disc - .as_object() - .ok_or_else(|| AlkTypeError::Schema("'discriminator' must be an object".to_string()))?; - let kind = disc_obj - .get("kind") - .and_then(Value::as_str) - .ok_or_else(|| AlkTypeError::Schema("discriminator is missing 'kind' field".to_string()))?; - match kind { - "byte" => { - let offset = disc_obj.get("offset").and_then(Value::as_u64).unwrap_or(0) as usize; - let disc_type_str = disc_obj - .get("type") - .and_then(Value::as_str) - .unwrap_or("AlkType:Uint8"); - let disc_type: AlkTypeKind = disc_type_str.parse().map_err(|_| { - AlkTypeError::Schema(format!( - "discriminator 'type' must be one of {BYTE_DISCRIMINATOR_TYPES:?}, got {disc_type_str:?}" - )) - })?; - if !BYTE_DISCRIMINATOR_TYPES.contains(&disc_type) { - return Err(AlkTypeError::Schema(format!( - "discriminator 'type' must be one of {BYTE_DISCRIMINATOR_TYPES:?}, got {disc_type:?}" - ))); - } - Ok(DiscriminatorKind::Byte { - offset, - disc_type, - }) - } - "field" => { - let name = disc_obj - .get("name") - .and_then(Value::as_str) - .ok_or_else(|| { - AlkTypeError::Schema( - "field discriminator is missing required 'name' field".to_string(), - ) - })? - .to_string(); - Ok(DiscriminatorKind::Field { name }) - } - other => Err(AlkTypeError::Schema(format!( - "unknown discriminator 'kind': {other:?} (expected \"byte\" or \"field\")" - ))), - } -} - -/// Detect a `AlkType:*` kind from a schema node, accepting either the -/// boolean form (`{ "AlkType:String": true }`) or the object-annotation -/// form (`{ "AlkType:String": { "encoding": "..." } }`). -/// -/// [`get_alktype_kind`] only recognizes the boolean form; layout computation -/// and engine dispatch also need to recognize the object form so that -/// variable-length encoding annotations don't hide the kind. -pub fn get_alktype_kind_loose(node: &Value) -> Option<&str> { - let obj = node.as_object()?; - for key in obj.keys() { - if key.starts_with(ALKTYPE_PREFIX) && obj.get(key).is_some_and(|v| !v.is_null()) { - return Some(key.as_str()); - } - } - None -} - -/// Like [`get_alktype_kind_loose`] but returns the parsed [`AlkTypeKind`] enum. -pub fn get_alktype_kind_loose_enum(node: &Value) -> Option { - get_alktype_kind_loose(node).and_then(|s| s.parse().ok()) -} - -/// Resolve a `$ref` against the root schema, or return the inline schema. -/// -/// If `node` has a `"$ref"` key, parse the JSON Pointer and walk `root`. -/// Otherwise, return `node` itself (it's an inline schema). -pub fn resolve_ref_or_inline<'a>(node: &'a Value, root: &'a Value) -> Option<&'a Value> { - let obj = node.as_object()?; - if let Some(Value::String(ref_path)) = obj.get("$ref") { - return resolve_ref(root, ref_path); - } - Some(node) -} - -/// Resolve a JSON Pointer `$ref` (e.g., `"#/$defs/Read"`) against `root`. -pub fn resolve_ref<'a>(root: &'a Value, ref_path: &str) -> Option<&'a Value> { - let stripped = ref_path.strip_prefix('#').unwrap_or(ref_path); - let stripped = stripped.strip_prefix('/').unwrap_or(stripped); - if stripped.is_empty() { - return Some(root); - } - let mut current = root; - for segment in stripped.split('/') { - let decoded = segment.replace("~1", "/").replace("~0", "~"); - if let Ok(idx) = decoded.parse::() { - current = current.get(idx)?; - } else { - current = current.get(&decoded)?; - } - } - Some(current) -} - -/// Walk the schema tree. For every `"$ref"` whose value is a bare name -/// (no `#` prefix), rewrite it to `"#/$defs/"`. Full JSON Pointer refs -/// (starting with `#`) pass through unchanged. Idempotent. -pub fn normalize_refs(schema: &mut Value) { - normalize_refs_recursive(schema); -} - -fn normalize_refs_recursive(node: &mut Value) { - if let Value::Object(obj) = node { - if let Some(Value::String(ref s)) = obj.get("$ref") { - if !s.starts_with('#') && !s.is_empty() { - let new_ref = format!("#/$defs/{s}"); - if let Some(slot) = obj.get_mut("$ref") { - *slot = Value::String(new_ref); - } - } - } - for value in obj.values_mut() { - normalize_refs_recursive(value); - } - } else if let Value::Array(arr) = node { - for item in arr.iter_mut() { - normalize_refs_recursive(item); - } - } -} - -/// Inline `$ref`s in `AlkType:Union` `mapping` entries by resolving them -/// against the schema root and replacing each `$ref`-bearing variant with -/// the resolved schema (in place). This runs after [`normalize_refs`] and -/// before [`crate::validation::build_validator`]. -/// -/// The `UnionValidator`'s factory builds a sub-validator for each variant -/// at validator-construction time. The factory receives the union node as -/// `parent`, but `$defs` live at the schema root — not on the union node. -/// Without inlining, the factory can't resolve `$ref`s like -/// `"#/$defs/Init"` because the union node doesn't contain `$defs`. -/// Inlining the refs before validator construction sidesteps this: each -/// variant in the `mapping` becomes a full inline schema, so the factory -/// can build a sub-validator directly. -/// -/// Only `AlkType:Union` mappings are inlined. Other `$ref`s (e.g. in -/// `properties` or `items`) are left in place — the materializer resolves -/// them at read time via [`resolve_ref_or_inline`] against the schema -/// root, and the jsonschema built-in validator resolves them via its own -/// `$ref` resolution (which has access to the full schema root). -/// -/// Idempotent: a mapping whose variants are already inline (no `$ref`) -/// is left unchanged. -pub fn inline_union_variant_refs(root: &mut Value) { - // Collect (path-to-union-mapping, ref-path, variant-key) tuples by - // walking the tree with immutable access, then resolve and apply - // the inlining with mutable access. This avoids the borrow conflict - // of holding both &root and &mut root simultaneously. - let root_clone = root.clone(); - inline_union_variant_refs_recursive(root, &root_clone); -} - -fn inline_union_variant_refs_recursive(node: &mut Value, root: &Value) { - if let Value::Object(obj) = node { - // Check if this node is an AlkType:Union with a mapping. - let is_union = obj - .keys() - .any(|k| k == "AlkType:Union"); - if is_union { - if let Some(Value::Object(mapping)) = obj.get_mut("mapping") { - for (_key, variant) in mapping.iter_mut() { - if let Some(Value::String(ref_path)) = variant.get("$ref").cloned() { - if let Some(resolved) = resolve_ref(root, &ref_path) { - *variant = resolved.clone(); - } - } - } - } - } - for value in obj.values_mut() { - inline_union_variant_refs_recursive(value, root); - } - } else if let Value::Array(arr) = node { - for item in arr.iter_mut() { - inline_union_variant_refs_recursive(item, root); - } - } -} - #[cfg(test)] mod tests { use super::*; - use serde_json::json; - - #[test] - fn get_alktype_kind_detects_bool_keyword() { - let schema = json!({"AlkType:Uint32": true}); - assert_eq!(get_alktype_kind(&schema), Some("AlkType:Uint32")); - } - - #[test] - fn get_alktype_kind_ignores_object_keyword() { - let schema = json!({"AlkType:String": {"encoding": "length-prefixed"}}); - assert_eq!(get_alktype_kind(&schema), None); - } - - #[test] - fn get_alktype_kind_none_for_plain_schema() { - let schema = json!({"type": "object", "properties": {}}); - assert_eq!(get_alktype_kind(&schema), None); - } #[test] fn type_size_fixed_kinds() { @@ -638,17 +257,10 @@ mod tests { AlkTypeKind::Record, AlkTypeKind::Timestamp, ] { - assert_eq!(kind.type_size(), None, "failed for {kind}"); + assert_eq!(kind.type_size(), None, "failed for {kind:?}"); } } - #[test] - fn type_size_unknown_kind_returns_none() { - assert!("AlkType:Uint128".parse::().is_err()); - assert!("AlkType:Int128".parse::().is_err()); - assert!("not-an-alktype".parse::().is_err()); - } - #[test] fn natural_alignment_matches_spec() { assert_eq!(AlkTypeKind::Int8.natural_alignment(), 1); @@ -688,7 +300,7 @@ mod tests { AlkTypeKind::Boolean, AlkTypeKind::Enum, ] { - assert!(kind.is_fixed_size(), "expected fixed: {kind}"); + assert!(kind.is_fixed_size(), "expected fixed: {kind:?}"); } for kind in [ AlkTypeKind::String, @@ -699,381 +311,10 @@ mod tests { AlkTypeKind::Record, AlkTypeKind::Timestamp, ] { - assert!(!kind.is_fixed_size(), "expected variable: {kind}"); + assert!(!kind.is_fixed_size(), "expected variable: {kind:?}"); } } - #[test] - fn endian_from_schema_defaults_to_little() { - assert_eq!(Endian::from_schema(&json!({})), Endian::Little); - assert_eq!( - Endian::from_schema(&json!({"endian": "little"})), - Endian::Little - ); - assert_eq!( - Endian::from_schema(&json!({"endian": "weird"})), - Endian::Little - ); - } - - #[test] - fn endian_from_schema_big() { - assert_eq!(Endian::from_schema(&json!({"endian": "big"})), Endian::Big); - } - - #[test] - fn parse_encoding_shorthand_true() { - assert_eq!( - parse_encoding(&json!(true)), - VariableEncoding::LengthPrefixed - ); - } - - #[test] - fn parse_encoding_object_length_prefixed() { - assert_eq!( - parse_encoding(&json!({"encoding": "length-prefixed"})), - VariableEncoding::LengthPrefixed - ); - } - - #[test] - fn parse_encoding_object_offset_indirect() { - assert_eq!( - parse_encoding(&json!({"encoding": "offset-indirect"})), - VariableEncoding::OffsetIndirect - ); - } - - #[test] - fn parse_encoding_unknown_defaults_to_length_prefixed() { - assert_eq!( - parse_encoding(&json!({"encoding": "weird"})), - VariableEncoding::LengthPrefixed - ); - assert_eq!(parse_encoding(&json!(42)), VariableEncoding::LengthPrefixed); - assert_eq!( - parse_encoding(&json!(null)), - VariableEncoding::LengthPrefixed - ); - } - - #[test] - fn parse_align_returns_value() { - assert_eq!(parse_align(&json!({"align": 256})), Some(256)); - assert_eq!(parse_align(&json!({"align": 0})), Some(0)); - } - - #[test] - fn parse_align_none_when_absent() { - assert_eq!(parse_align(&json!({})), None); - assert_eq!(parse_align(&json!({"align": "not-a-number"})), None); - } - - #[test] - fn parse_max_length_returns_value() { - assert_eq!(parse_max_length(&json!({"maxLength": 1024})), Some(1024)); - } - - #[test] - fn parse_max_length_none_when_absent() { - assert_eq!(parse_max_length(&json!({})), None); - assert_eq!(parse_max_length(&json!({"maxLength": "x"})), None); - } - - #[test] - fn parse_endian_alias_matches_from_schema() { - assert_eq!(parse_endian(&json!({"endian": "big"})), Endian::Big); - assert_eq!(parse_endian(&json!({})), Endian::Little); - } - - #[test] - fn parse_discriminator_byte_default_offset_and_type() { - let schema = json!({"discriminator": {"kind": "byte"}}); - let disc = parse_discriminator(&schema).expect("byte discriminator"); - assert_eq!( - disc, - DiscriminatorKind::Byte { - offset: 0, - disc_type: AlkTypeKind::Uint8, - } - ); - } - - #[test] - fn parse_discriminator_byte_explicit() { - let schema = json!({ - "discriminator": {"kind": "byte", "offset": 4, "type": "AlkType:Uint16"} - }); - let disc = parse_discriminator(&schema).expect("byte discriminator"); - assert_eq!( - disc, - DiscriminatorKind::Byte { - offset: 4, - disc_type: AlkTypeKind::Uint16, - } - ); - } - - #[test] - fn parse_discriminator_field() { - let schema = json!({"discriminator": {"kind": "field", "name": "type"}}); - let disc = parse_discriminator(&schema).expect("field discriminator"); - assert_eq!( - disc, - DiscriminatorKind::Field { - name: "type".to_string() - } - ); - } - - #[test] - fn parse_discriminator_missing_discriminator_is_error() { - let schema = json!({"AlkType:Union": true}); - assert!(matches!( - parse_discriminator(&schema), - Err(AlkTypeError::Schema(_)) - )); - } - - #[test] - fn parse_discriminator_field_missing_name_is_error() { - let schema = json!({"discriminator": {"kind": "field"}}); - assert!(matches!( - parse_discriminator(&schema), - Err(AlkTypeError::Schema(_)) - )); - } - - #[test] - fn parse_discriminator_unknown_kind_is_error() { - let schema = json!({"discriminator": {"kind": "magic"}}); - assert!(matches!( - parse_discriminator(&schema), - Err(AlkTypeError::Schema(_)) - )); - } - - #[test] - fn parse_discriminator_byte_invalid_type_is_error() { - let schema = json!({ - "discriminator": {"kind": "byte", "type": "AlkType:Float32"} - }); - assert!(matches!( - parse_discriminator(&schema), - Err(AlkTypeError::Schema(_)) - )); - } - - #[test] - fn normalize_refs_rewrites_bare_name() { - let mut schema = json!({"$ref": "Read"}); - normalize_refs(&mut schema); - assert_eq!(schema, json!({"$ref": "#/$defs/Read"})); - } - - #[test] - fn normalize_refs_leaves_pointer_ref_unchanged() { - let mut schema = json!({"$ref": "#/$defs/Read"}); - normalize_refs(&mut schema); - assert_eq!(schema, json!({"$ref": "#/$defs/Read"})); - } - - #[test] - fn normalize_refs_is_idempotent() { - let mut schema = json!({"$ref": "Read"}); - normalize_refs(&mut schema); - normalize_refs(&mut schema); - assert_eq!(schema, json!({"$ref": "#/$defs/Read"})); - } - - #[test] - fn normalize_refs_walks_nested_objects() { - let mut schema = json!({ - "properties": { - "child": {"$ref": "Child"}, - "other": {"$ref": "#/$defs/Other"} - }, - "items": [ - {"$ref": "InArray"}, - {"foo": {"$ref": "Deep"}} - ] - }); - normalize_refs(&mut schema); - assert_eq!( - schema, - json!({ - "properties": { - "child": {"$ref": "#/$defs/Child"}, - "other": {"$ref": "#/$defs/Other"} - }, - "items": [ - {"$ref": "#/$defs/InArray"}, - {"foo": {"$ref": "#/$defs/Deep"}} - ] - }) - ); - } - - #[test] - fn normalize_refs_preserves_sibling_keys() { - let mut schema = json!({ - "$ref": "Read", - "alktype:annotation": "kept" - }); - normalize_refs(&mut schema); - assert_eq!( - schema, - json!({ - "$ref": "#/$defs/Read", - "alktype:annotation": "kept" - }) - ); - } - - // ----- inline_union_variant_refs tests ----- - - #[test] - fn inline_union_variant_refs_inlines_mapping_refs() { - let mut schema = json!({ - "AlkType:Struct": true, - "properties": { - "payload": { - "AlkType:Union": true, - "discriminator": { "kind": "byte", "offset": 0, "type": "AlkType:Uint8" }, - "mapping": { - "5": {"$ref": "#/$defs/Read"}, - "6": {"$ref": "#/$defs/Write"} - } - } - }, - "$defs": { - "Read": { "AlkType:Struct": true, "properties": { "id": { "AlkType:Uint32": true } } }, - "Write": { "AlkType:Struct": true, "properties": { "n": { "AlkType:Uint16": true } } } - } - }); - inline_union_variant_refs(&mut schema); - // The mapping entries should now be the full inline schemas. - let mapping = &schema["properties"]["payload"]["mapping"]; - assert_eq!( - mapping["5"], - json!({ "AlkType:Struct": true, "properties": { "id": { "AlkType:Uint32": true } } }) - ); - assert_eq!( - mapping["6"], - json!({ "AlkType:Struct": true, "properties": { "n": { "AlkType:Uint16": true } } }) - ); - // $defs are left in place (the materializer may still use them). - assert!(schema["$defs"].is_object()); - } - - #[test] - fn inline_union_variant_refs_leaves_inline_variants_unchanged() { - let mut schema = json!({ - "AlkType:Union": true, - "discriminator": { "kind": "byte", "offset": 0, "type": "AlkType:Uint8" }, - "mapping": { - "5": { "AlkType:Struct": true, "properties": { "id": { "AlkType:Uint32": true } } } - } - }); - let before = schema.clone(); - inline_union_variant_refs(&mut schema); - assert_eq!(schema, before, "inline variants should be unchanged"); - } - - #[test] - fn inline_union_variant_refs_is_idempotent() { - let mut schema = json!({ - "AlkType:Union": true, - "discriminator": { "kind": "byte", "offset": 0, "type": "AlkType:Uint8" }, - "mapping": { "5": {"$ref": "#/$defs/Read"} }, - "$defs": { "Read": { "AlkType:Struct": true, "properties": { "id": { "AlkType:Uint32": true } } } } - }); - inline_union_variant_refs(&mut schema); - let after_first = schema.clone(); - inline_union_variant_refs(&mut schema); - assert_eq!(schema, after_first, "second inlining should be a no-op"); - } - - #[test] - fn inline_union_variant_refs_handles_nested_union() { - // A struct containing a union whose variants are $refs. The - // inlining walks into the struct's properties and inlines the - // union's mapping refs. - let mut schema = json!({ - "AlkType:Struct": true, - "properties": { - "outer": { - "AlkType:Struct": true, - "properties": { - "inner_union": { - "AlkType:Union": true, - "discriminator": { "kind": "byte", "offset": 0, "type": "AlkType:Uint8" }, - "mapping": { "1": {"$ref": "#/$defs/Init"} } - } - } - } - }, - "$defs": { "Init": { "AlkType:Struct": true, "properties": { "v": { "AlkType:Uint32": true } } } } - }); - inline_union_variant_refs(&mut schema); - let mapping = &schema["properties"]["outer"]["properties"]["inner_union"]["mapping"]; - assert_eq!( - mapping["1"], - json!({ "AlkType:Struct": true, "properties": { "v": { "AlkType:Uint32": true } } }) - ); - } - - #[test] - fn inline_union_variant_refs_leaves_non_union_refs_in_place() { - // $refs in `properties` (not in a union mapping) are left in - // place — the materializer resolves them at read time. - let mut schema = json!({ - "AlkType:Struct": true, - "properties": { - "field": {"$ref": "#/$defs/SomeType"} - }, - "$defs": { "SomeType": { "AlkType:Uint32": true } } - }); - let before = schema.clone(); - inline_union_variant_refs(&mut schema); - assert_eq!(schema, before, "non-union refs should be unchanged"); - } - - #[test] - fn as_str_round_trips_for_all_kinds() { - for kind in [ - AlkTypeKind::Int8, - AlkTypeKind::Int16, - AlkTypeKind::Int32, - AlkTypeKind::Int64, - AlkTypeKind::Uint8, - AlkTypeKind::Uint16, - AlkTypeKind::Uint32, - AlkTypeKind::Uint64, - AlkTypeKind::Float32, - AlkTypeKind::Float64, - AlkTypeKind::Boolean, - AlkTypeKind::Enum, - AlkTypeKind::String, - AlkTypeKind::Bytes, - AlkTypeKind::Struct, - AlkTypeKind::Union, - AlkTypeKind::Array, - AlkTypeKind::Record, - AlkTypeKind::Timestamp, - ] { - let s = kind.as_str(); - assert_eq!(s.parse::().unwrap(), kind, "{kind}"); - } - } - - #[test] - fn display_uses_as_str() { - assert_eq!(format!("{}", AlkTypeKind::Uint32), "AlkType:Uint32"); - assert_eq!(format!("{}", AlkTypeKind::String), "AlkType:String"); - } - #[test] fn needs_endian_classifies_correctly() { for kind in [ @@ -1090,7 +331,7 @@ mod tests { AlkTypeKind::Bytes, AlkTypeKind::Timestamp, ] { - assert!(kind.needs_endian(), "expected needs_endian: {kind}"); + assert!(kind.needs_endian(), "expected needs_endian: {kind:?}"); } for kind in [ AlkTypeKind::Int8, @@ -1101,7 +342,7 @@ mod tests { AlkTypeKind::Array, AlkTypeKind::Record, ] { - assert!(!kind.needs_endian(), "expected not needs_endian: {kind}"); + assert!(!kind.needs_endian(), "expected not needs_endian: {kind:?}"); } } @@ -1113,7 +354,7 @@ mod tests { AlkTypeKind::Array, AlkTypeKind::Record, ] { - assert!(kind.is_composite(), "expected composite: {kind}"); + assert!(kind.is_composite(), "expected composite: {kind:?}"); } for kind in [ AlkTypeKind::Int8, @@ -1132,22 +373,10 @@ mod tests { AlkTypeKind::Bytes, AlkTypeKind::Timestamp, ] { - assert!(!kind.is_composite(), "expected not composite: {kind}"); + assert!(!kind.is_composite(), "expected not composite: {kind:?}"); } } - #[test] - fn get_alktype_kind_enum_returns_none_for_object_form() { - let schema = json!({"AlkType:String": {"encoding": "length-prefixed"}}); - assert_eq!(get_alktype_kind_enum(&schema), None); - } - - #[test] - fn get_alktype_kind_enum_returns_kind_for_bool_form() { - let schema = json!({"AlkType:Uint32": true}); - assert_eq!(get_alktype_kind_enum(&schema), Some(AlkTypeKind::Uint32)); - } - #[test] fn to_bast_str_covers_all_19_variants() { let cases: &[(AlkTypeKind, &str)] = &[ @@ -1233,18 +462,4 @@ mod tests { assert!(AlkTypeKind::from_bast_str("AlkType:Uint32").is_err()); assert!(AlkTypeKind::from_bast_str("").is_err()); } - - #[test] - fn from_bast_str_distinct_from_fromstr_keyword_form() { - assert!("AlkType:Uint32".parse::().is_ok()); - assert!(AlkTypeKind::from_bast_str("AlkType:Uint32").is_err()); - assert!(AlkTypeKind::from_bast_str("uint32").is_ok()); - assert!("uint32".parse::().is_err()); - } - - #[test] - fn to_bast_str_distinct_from_as_str() { - assert_eq!(AlkTypeKind::Uint32.as_str(), "AlkType:Uint32"); - assert_eq!(AlkTypeKind::Uint32.to_bast_str(), "uint32"); - } -} +} \ No newline at end of file