diff --git a/canonical_test.go b/canonical_test.go
index ddcea66..12fdc7d 100644
--- a/canonical_test.go
+++ b/canonical_test.go
@@ -1,6 +1,8 @@
package walle
import (
+ "encoding/json"
+ "fmt"
"reflect"
"strings"
"testing"
@@ -200,7 +202,12 @@ func TestCanonicalWithAutoFix(t *testing.T) {
}`,
},
{
- name: "invalid_type_with_ref",
+ // "$ref":"#" points at a root typed as object while the sibling asks for a
+ // string, so nothing can satisfy both. Dropping only the sibling type used
+ // to leave "$ref":"#" behind, which retypes the field to object and
+ // inverts what the schema asked for. Degrading to {} keeps the field
+ // unconstrained instead of quietly asserting the opposite type.
+ name: "unsatisfiable_type_with_ref",
invalidSchema: `{
"type": "object",
"properties": {
@@ -213,9 +220,7 @@ func TestCanonicalWithAutoFix(t *testing.T) {
simplifiedSchema: `{
"type": "object",
"properties": {
- "user": {
- "$ref": "#"
- }
+ "user": {}
}
}`,
},
@@ -483,6 +488,8 @@ func TestCanonicalWithAutoFix(t *testing.T) {
}`,
},
{
+ // Only the unusable entry goes; "name" is declared and genuinely required,
+ // so releasing it along with the empty string would loosen the schema.
name: "property_names_in_required_array_cannot_be_empty",
invalidSchema: `{
"type": "object",
@@ -495,7 +502,8 @@ func TestCanonicalWithAutoFix(t *testing.T) {
"type": "object",
"properties": {
"name": {"type": "string"}
- }
+ },
+ "required": ["name"]
}`,
},
{
@@ -559,13 +567,13 @@ func TestCanonicalWithAutoFix(t *testing.T) {
"start_url"
]
}`,
+ // The outer "type":["null"] applies on top of whichever branch matches, so
+ // the string branch can never be reached and is dropped. Keeping it would
+ // let the property accept strings that the outer type ruled out.
simplifiedSchema: `{
"properties": {
"start_url": {
"anyOf": [
- {
- "type": "string"
- },
{
"type": "null"
}
@@ -967,6 +975,9 @@ func TestCanonicalWithAutoFix(t *testing.T) {
// }`,
// },
{
+ // The outer minLength applies on top of whichever branch matches, so it
+ // is pushed into both; the branch asking for 10 keeps the stricter 20.
+ // The outer description carries no constraint and is simply dropped.
name: "conflicting_keywords_expand_anyOf_0",
invalidSchema: `{
"description": "xxx",
@@ -984,7 +995,20 @@ func TestCanonicalWithAutoFix(t *testing.T) {
}
]
}`,
- simplifiedSchema: `{}`,
+ simplifiedSchema: `{
+ "anyOf": [
+ {
+ "description": "yyy",
+ "type": "string",
+ "minLength": 20
+ },
+ {
+ "description": "zzz",
+ "type": "string",
+ "minLength": 20
+ }
+ ]
+ }`,
},
{
name: "conflicting_keywords_expand_anyOf_1",
@@ -1012,7 +1036,20 @@ func TestCanonicalWithAutoFix(t *testing.T) {
simplifiedSchema: `{
"type": "object",
"properties": {
- "name": {}
+ "name": {
+ "anyOf": [
+ {
+ "description": "yyy",
+ "type": "string",
+ "minLength": 20
+ },
+ {
+ "description": "zzz",
+ "type": "string",
+ "minLength": 20
+ }
+ ]
+ }
}
}`,
},
@@ -1296,7 +1333,10 @@ func TestCanonicalCommonKeywordConflictSimplify(t *testing.T) {
}`,
},
{
- name: "mixed_structural_conflict_still_clears_parent_sub_schema",
+ // A constraint and an annotation clashing at once are handled one at a
+ // time: the constraint is pushed into the branch with the stricter value
+ // winning, and the annotation's outer copy is dropped.
+ name: "mixed_structural_conflict_distributes_and_drops_the_annotation",
invalidSchema: `{
"type": "object",
"properties": {
@@ -1312,7 +1352,11 @@ func TestCanonicalCommonKeywordConflictSimplify(t *testing.T) {
simplifiedSchema: `{
"type": "object",
"properties": {
- "name": {}
+ "name": {
+ "anyOf": [
+ { "type": "string", "description": "yyy", "minLength": 20 }
+ ]
+ }
}
}`,
},
@@ -1638,7 +1682,10 @@ func TestCanonicalRemovesSchemaKeywordOnly(t *testing.T) {
}
}
-func TestCanonicalDoesNotSimplifyOtherUnsupportedKeywordsWithSchema(t *testing.T) {
+// Unsupported keywords are dropped individually rather than collapsing the whole
+// schema: the enforcer skips keys it does not recognise, so dropping the key alone
+// yields the same constrained decoding while keeping every sibling constraint.
+func TestCanonicalRemovesUnsupportedKeywordsInsteadOfWiping(t *testing.T) {
schema, err := ParseSchema(`{
"type": "object",
"$schema": "https://json-schema.org/draft/2020-12/schema",
@@ -1659,7 +1706,463 @@ func TestCanonicalDoesNotSimplifyOtherUnsupportedKeywordsWithSchema(t *testing.T
if !strings.Contains(warnErr.Error(), "unsupported keywords: $schema, zzz") {
t.Fatalf("expected warning to include both unsupported keywords, got %v", warnErr)
}
- if result != "{}" {
- t.Fatalf("expected other unsupported keywords to keep original fallback behavior, got %s", result)
+ if result == "{}" {
+ t.Fatal("expected schema to be preserved, got empty object")
+ }
+
+ fixed, err := ParseSchema(result)
+ if err != nil {
+ t.Fatalf("Fixed schema is not valid JSON: %v", err)
+ }
+ for _, k := range []string{"$schema", "zzz"} {
+ if _, ok := fixed[k]; ok {
+ t.Errorf("unsupported keyword %q should be removed", k)
+ }
+ }
+ if fixed["type"] != "object" {
+ t.Errorf("type should remain, got %v", fixed["type"])
+ }
+ props, ok := fixed["properties"].(map[string]any)
+ if !ok || props["command"] == nil {
+ t.Error("properties.command should remain")
+ }
+ required, ok := fixed["required"].([]any)
+ if !ok || len(required) != 1 || required[0] != "command" {
+ t.Errorf("required should remain, got %v", fixed["required"])
+ }
+}
+
+// Canonical must degrade the offending subschema, never the whole document.
+// Each case below used to collapse to "{}" because its error carried
+// SimplifyDefault, which is a no-op: the schema never changed, the same error
+// resurfaced every round, and the retry loop fell through to the empty fallback.
+func TestCanonicalDegradesLocallyNotWholeSchema(t *testing.T) {
+ cases := []struct {
+ name string
+ in string
+ // keep: JSON fragments that must survive; gone: fragments that must not
+ keep []string
+ gone []string
+ }{
+ {
+ name: "unsupported keyword on a property",
+ in: `{"type":"object","properties":{"a":{"type":"string","format":"uuid"}}}`,
+ keep: []string{`"type":"string"`},
+ gone: []string{"format"},
+ },
+ {
+ name: "unsupported keyword at root",
+ in: `{"type":"object","properties":{"a":{"type":"string"}},"required":["a"],"allOf":[{"title":"x"}]}`,
+ keep: []string{`"required":["a"]`, `"type":"string"`},
+ gone: []string{"allOf"},
+ },
+ {
+ // The sibling constraint is folded into the definition instead of being
+ // deleted, so nothing the schema asked for is lost.
+ name: "sibling constraint next to $ref is merged in",
+ in: `{"type":"object","properties":{"t":{"maxLength":9,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","minLength":1}}}`,
+ keep: []string{`"maxLength":9`, `"minLength":1`, `"type":"string"`},
+ gone: []string{`"$ref"`},
+ },
+ {
+ name: "$defs outside root",
+ in: `{"type":"object","properties":{"a":{"type":"object","$defs":{"X":{"type":"string"}},"properties":{"b":{"type":"string"}}}}}`,
+ keep: []string{`"b":{"type":"string"}`},
+ gone: []string{"$defs"},
+ },
+ {
+ // The empty string is a property name like any other, and the enforcer
+ // generates it, so nothing about this node needs rewriting.
+ name: "empty property name",
+ in: `{"type":"object","properties":{"":{"type":"string"},"a":{"type":"number"}}}`,
+ keep: []string{`"":{"type":"string"}`, `"a":{"type":"number"}`},
+ },
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ schema, err := ParseSchema(tc.in)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+
+ result, _ := schema.Canonical()
+ if result == "{}" {
+ t.Fatalf("schema collapsed to {}; expected only the offending part to be dropped")
+ }
+ for _, frag := range tc.keep {
+ if !strings.Contains(result, frag) {
+ t.Errorf("expected %s to survive, got %s", frag, result)
+ }
+ }
+ for _, frag := range tc.gone {
+ if strings.Contains(result, frag) {
+ t.Errorf("expected %s to be dropped, got %s", frag, result)
+ }
+ }
+ })
+ }
+}
+
+// A keyword next to $ref is an independent assertion, so the effective
+// constraint is the conjunction of both sides. Canonical folds the definition
+// into the use site and keeps whichever value admits fewer instances, rather
+// than deleting the sibling and silently accepting whatever the definition says.
+func TestCanonicalMergesRefSiblingsKeepingStricterValue(t *testing.T) {
+ cases := []struct {
+ name string
+ in string
+ keep []string
+ gone []string
+ }{
+ {
+ name: "lower bound keeps the larger value",
+ in: `{"type":"object","properties":{"t":{"minLength":5,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","minLength":1}}}`,
+ keep: []string{`"minLength":5`},
+ gone: []string{`"minLength":1`},
+ },
+ {
+ name: "lower bound keeps the larger value when the definition is stricter",
+ in: `{"type":"object","properties":{"t":{"minLength":1,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","minLength":5}}}`,
+ keep: []string{`"minLength":5`},
+ gone: []string{`"minLength":1`},
+ },
+ {
+ name: "upper bound keeps the smaller value",
+ in: `{"type":"object","properties":{"t":{"maxLength":9,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","maxLength":3}}}`,
+ keep: []string{`"maxLength":3`},
+ gone: []string{`"maxLength":9`},
+ },
+ {
+ name: "a constraint the definition lacks is preserved",
+ in: `{"type":"object","properties":{"t":{"maxLength":9,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","minLength":1}}}`,
+ keep: []string{`"maxLength":9`, `"minLength":1`, `"type":"string"`},
+ },
+ {
+ name: "type narrows to the intersection",
+ in: `{"type":"object","properties":{"t":{"type":"integer","minimum":3,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"number"}}}`,
+ keep: []string{`"type":"integer"`, `"minimum":3`},
+ gone: []string{`"type":"number"`},
+ },
+ {
+ name: "enum narrows to the intersection",
+ in: `{"type":"object","properties":{"t":{"enum":["a","b"],"minLength":1,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","enum":["b","c"]}}}`,
+ keep: []string{`"enum":["b"]`},
+ },
+ {
+ name: "required takes the union",
+ in: `{"type":"object","properties":{"t":{"required":["x"],"minLength":1,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"object","properties":{"x":{"type":"string"},"y":{"type":"string"}},"required":["y"]}}}`,
+ keep: []string{`"x"`, `"y"`},
+ },
+ {
+ // Two uses of one definition get their own merged copy, so neither
+ // bound leaks into the other.
+ name: "each use site is specialised independently",
+ in: `{"type":"object","properties":{"a":{"minLength":5,"$ref":"#/$defs/S"},"b":{"minLength":9,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","minLength":1}}}`,
+ keep: []string{`"a":{"minLength":5`, `"b":{"minLength":9`},
+ },
+ {
+ // Inlining a recursive definition would not terminate, so the sibling
+ // is dropped as before and the reference stays.
+ name: "recursive definition falls back to dropping the sibling",
+ in: `{"type":"object","properties":{"t":{"minLength":5,"$ref":"#/$defs/N"}},"$defs":{"N":{"type":"object","properties":{"next":{"$ref":"#/$defs/N"}}}}}`,
+ keep: []string{`"$ref":"#/$defs/N"`},
+ gone: []string{`"minLength"`},
+ },
+ {
+ // Nothing would be lost by dropping the duplicate, so the reference is
+ // left in place and stays shared.
+ name: "identical values keep the reference",
+ in: `{"type":"object","properties":{"t":{"minLength":1,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","minLength":1}}}`,
+ keep: []string{`"$ref":"#/$defs/S"`},
+ },
+ {
+ name: "annotation-only siblings keep the reference",
+ in: `{"type":"object","properties":{"t":{"description":"outer","$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","description":"inner"}}}`,
+ keep: []string{`"$ref":"#/$defs/S"`},
+ },
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ schema, err := ParseSchema(tc.in)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+
+ result, _ := schema.Canonical()
+ if result == "{}" {
+ t.Fatalf("schema collapsed to {}")
+ }
+ for _, frag := range tc.keep {
+ if !strings.Contains(result, frag) {
+ t.Errorf("expected %s in result, got %s", frag, result)
+ }
+ }
+ for _, frag := range tc.gone {
+ if strings.Contains(result, frag) {
+ t.Errorf("expected %s to be gone, got %s", frag, result)
+ }
+ }
+ })
+ }
+}
+
+// An unsatisfiable overlap must not be simplified by deleting one of the two
+// sides: whichever side survives, the node ends up accepting exactly what the
+// original rejected. Only the contradicting keyword goes, so the node keeps
+// whatever the two sides still agree on and is left unconstrained in that one
+// dimension.
+//
+// Reporting order matters here: the rules that delete $ref siblings to
+// canonicalise a node used to run first, and once the contradicting sibling was
+// gone nothing was left to notice the contradiction.
+func TestCanonicalDoesNotSimplifyAwayUnsatisfiableRefSiblings(t *testing.T) {
+ cases := []struct {
+ name string
+ in string
+ // wantProperty is the complete canonical form of the contradicting
+ // property, so an inverted result cannot slip past
+ wantProperty string
+ }{
+ {
+ // Nothing survives an empty type intersection, so the property is left
+ // fully unconstrained rather than typed as number.
+ name: "contradicting type",
+ in: `{"type":"object","properties":{"t":{"type":"string","$ref":"#/$defs/S"}},"$defs":{"S":{"type":"number"}}}`,
+ wantProperty: `"t":{}`,
+ },
+ {
+ // Constraints either side carries cannot be salvaged once the types
+ // disagree: a string length and a numeric minimum in one typeless node
+ // describe nothing at all.
+ name: "contradicting type discards both sides' constraints",
+ in: `{"type":"object","properties":{"t":{"type":"string","minLength":3,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"number","minimum":5}}}`,
+ wantProperty: `"t":{}`,
+ },
+ {
+ // Both sides agree the value is a string and only disagree on which
+ // strings, so the type is kept and only the enum is dropped. Keeping
+ // either enum would accept a value the other side ruled out.
+ name: "contradicting enum keeps the agreed type",
+ in: `{"type":"object","properties":{"t":{"type":"string","enum":["x"],"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","enum":["y"]}}}`,
+ wantProperty: `"t":{"type":"string"}`,
+ },
+ {
+ // Dropping the enum must not drop everything else the two sides agreed
+ // on, and what survives still has to be the stricter of the two bounds.
+ name: "contradicting enum keeps the stricter bound",
+ in: `{"type":"object","properties":{"t":{"type":"string","enum":["x"],"minLength":2,"$ref":"#/$defs/S"}},"$defs":{"S":{"type":"string","enum":["y"],"minLength":9}}}`,
+ wantProperty: `"t":{"minLength":9,"type":"string"}`,
+ },
+ {
+ // A recursive definition cannot be inlined, so there is no merged form
+ // to keep and the node is emptied instead of adopting the object type.
+ name: "contradicting type against a recursive definition",
+ in: `{"type":"object","properties":{"t":{"type":"string","$ref":"#/$defs/S"}},"$defs":{"S":{"type":"object","properties":{"next":{"$ref":"#/$defs/S"}}}}}`,
+ wantProperty: `"t":{}`,
+ },
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ schema, err := ParseSchema(tc.in)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+
+ result, warnErr := schema.Canonical()
+ if warnErr == nil {
+ t.Fatal("expected a warning about the empty intersection")
+ }
+ if !strings.Contains(warnErr.Error(), "intersection is empty") {
+ t.Fatalf("expected an empty-intersection warning, got %v", warnErr)
+ }
+ if !strings.Contains(result, tc.wantProperty) {
+ t.Fatalf("expected the property to become %s, got %s", tc.wantProperty, result)
+ }
+ })
+ }
+}
+
+// Both sides describing the same object must end up with both sets of
+// properties, and a property both sides describe keeps both sides' constraints.
+// Letting the use site win outright would quietly drop whatever the definition
+// asserted about that property.
+func TestCanonicalMergesPropertiesFromBothSides(t *testing.T) {
+ const schema = `{
+ "type":"object",
+ "properties":{
+ "node":{
+ "properties":{"shared":{"type":"string"},"onlyHere":{"type":"boolean"}},
+ "$ref":"#/$defs/S"
+ }
+ },
+ "$defs":{
+ "S":{
+ "type":"object",
+ "properties":{"shared":{"type":"string","minLength":10},"onlyThere":{"type":"integer"}}
+ }
+ }
+ }`
+
+ parsed, err := ParseSchema(schema)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ result, _ := parsed.Canonical()
+
+ var out map[string]any
+ if err := json.Unmarshal([]byte(result), &out); err != nil {
+ t.Fatalf("canonical output is not valid JSON: %v", err)
+ }
+
+ properties := out["properties"].(map[string]any)["node"].(map[string]any)["properties"]
+ merged, ok := properties.(map[string]any)
+ if !ok {
+ t.Fatalf("expected merged properties, got %s", result)
+ }
+
+ for _, name := range []string{"shared", "onlyHere", "onlyThere"} {
+ if _, exists := merged[name]; !exists {
+ t.Errorf("expected property %q to survive the merge, got %s", name, result)
+ }
+ }
+
+ shared, ok := merged["shared"].(map[string]any)
+ if !ok {
+ t.Fatalf("expected shared to be a schema, got %s", result)
+ }
+ if shared["minLength"] != float64(10) {
+ t.Errorf("expected the definition's minLength to survive on shared, got %s", result)
+ }
+ if shared["type"] != "string" {
+ t.Errorf("expected the use site's type to survive on shared, got %s", result)
+ }
+}
+
+// type is intersected rather than replaced, and integer wins an integer/number
+// overlap because every integer is a number but not the reverse.
+func TestCanonicalIntersectsTypesWhenMerging(t *testing.T) {
+ cases := []struct {
+ name string
+ siblings string
+ target string
+ expectType string
+ }{
+ {
+ name: "integer against number keeps integer",
+ siblings: `"type":"number","minimum":5`,
+ target: `"type":"integer"`,
+ expectType: `"type":"integer"`,
+ },
+ {
+ name: "number against integer keeps integer",
+ siblings: `"type":"integer","minimum":5`,
+ target: `"type":"number"`,
+ expectType: `"type":"integer"`,
+ },
+ {
+ name: "overlapping unions keep the shared members",
+ siblings: `"type":["string","integer","boolean"],"description":"d"`,
+ target: `"type":["string","boolean"]`,
+ expectType: `"type":["boolean","string"]`,
+ },
+ {
+ name: "union narrowed to a single type collapses to a string",
+ siblings: `"type":["string","integer"],"minLength":2`,
+ target: `"type":["string"]`,
+ expectType: `"type":"string"`,
+ },
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ schema := fmt.Sprintf(
+ `{"type":"object","properties":{"v":{%s,"$ref":"#/$defs/S"}},"$defs":{"S":{%s}}}`,
+ tc.siblings, tc.target,
+ )
+ parsed, err := ParseSchema(schema)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ result, _ := parsed.Canonical()
+ if !strings.Contains(result, tc.expectType) {
+ t.Errorf("expected %s in the merged node, got %s", tc.expectType, result)
+ }
+ })
+ }
+}
+
+// Inlining $ref siblings copies the referenced definition at every use site, so
+// a chain of definitions referenced twice per level would double the output per
+// level -- ~900KB here for a 5KB input. The copy budget stops the inlining
+// before the size limit can, the siblings that could not be folded in are
+// dropped in one pass, and Canonical returns the loosened schema with a warning
+// that names what was lost -- not a {} from the size check.
+func TestCanonicalStopsInliningBeyondCopyBudget(t *testing.T) {
+ var sb strings.Builder
+ sb.WriteString(`{"type":"object","properties":{"root":{"$ref":"#/$defs/D1","description":"use site"}},"$defs":{`)
+ const levels = 14
+ for i := 1; i < levels; i++ {
+ if i > 1 {
+ sb.WriteString(",")
+ }
+ fmt.Fprintf(&sb, `"D%d":{"type":"object","properties":{"p":{"$ref":"#/$defs/D%d","additionalProperties":false},"q":{"$ref":"#/$defs/D%d","additionalProperties":false}}}`, i, i+1, i+1)
+ }
+ fmt.Fprintf(&sb, `,"D%d":{"type":"object","properties":{"x":{"type":"string"}}}}}`, levels)
+
+ schema, err := ParseSchema(sb.String())
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+
+ out, canonicalErr := schema.Canonical()
+ if canonicalErr == nil {
+ t.Fatal("expected a warning listing the dropped constraints")
+ }
+ if !strings.Contains(canonicalErr.Error(), "additionalProperties") {
+ t.Fatalf("warning should name the dropped keyword, got: %v", canonicalErr)
+ }
+ if out == "{}" {
+ t.Fatal("expected a loosened schema, got an empty one")
+ }
+ var parsed map[string]any
+ if err := json.Unmarshal([]byte(out), &parsed); err != nil {
+ t.Fatalf("canonical output is not valid JSON: %v", err)
+ }
+ // Constraints folded in before the budget ran out survive inlined; the ones
+ // that did not fit are gone from their $ref nodes.
+ if !strings.Contains(out, `"$ref"`) {
+ t.Fatalf("references should survive the drop: %s", out)
+ }
+ if !strings.Contains(out, `"additionalProperties":false`) {
+ t.Fatalf("constraints inlined before the budget ran out should survive: %s", out)
+ }
+}
+
+// Schemas well under the copy budget must still inline fully.
+func TestCanonicalInliningBelowCopyBudgetUnchanged(t *testing.T) {
+ raw := `{
+ "type": "object",
+ "properties": {
+ "short": {"maxLength": 10, "$ref": "#/$defs/S"},
+ "long": {"maxLength": 99, "$ref": "#/$defs/S"}
+ },
+ "$defs": {"S": {"type": "string", "minLength": 1}}
+ }`
+
+ schema, err := ParseSchema(raw)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ out, err := schema.Canonical()
+ if err != nil {
+ t.Fatalf("canonical failed: %v", err)
+ }
+ for _, want := range []string{`"maxLength":10`, `"maxLength":99`, `"minLength":1`} {
+ if !strings.Contains(out, want) {
+ t.Errorf("expected %s in canonical output, got %s", want, out)
+ }
}
}
diff --git a/docs/mfjs-walle-vs-draft-2020-12.md b/docs/mfjs-walle-vs-draft-2020-12.md
index ba67c3f..181ae0c 100644
--- a/docs/mfjs-walle-vs-draft-2020-12.md
+++ b/docs/mfjs-walle-vs-draft-2020-12.md
@@ -21,7 +21,7 @@ This document summarizes how the **walle** validator handles each keyword in **J
| --- | --- |
| `$id` | ✅ **Root only**; value must be a string. |
| `$schema` | ❌ |
-| `$ref` | ✅ **In-document references only**; remote / URL / cross-file references are disallowed; infinite recursion must be avoided. |
+| `$ref` | ✅ **In-document references only**; remote / URL / cross-file references are disallowed; infinite recursion must be avoided (the test is whether a finite instance exists); siblings of `$ref` apply as a logical AND, per 2020-12. |
| `$comment` | ❌ |
| `$defs` | ✅ **Root only**; **definition names must not contain `/`**. |
| `$anchor` | ❌ |
@@ -36,13 +36,13 @@ This document summarizes how the **walle** validator handles each keyword in **J
| Keyword | walle |
| --- | --- |
| `allOf` | ❌ |
-| `anyOf` | ✅ Branch count **may be capped**; `type` must **not** appear beside `anyOf` / `$ref` at the same level—declare `type` **inside** each branch. |
+| `anyOf` | ✅ Branch count **may be capped**; constraints such as `type` may sit beside it—they apply as a logical AND, and `Canonical` distributes them into every branch. |
| `oneOf` | ❌ |
| `if` | ❌ |
| `then` | ❌ |
| `else` | ❌ |
| `not` | ❌ |
-| `properties` | ✅ When `type` is `object`: **keys must not** be `$defs`, `$ref`, `anyOf`, `required`, or `additionalProperties`; **no duplicate keys**; every name in `required` must appear in `properties`. |
+| `properties` | ✅ When `type` is `object`: **keys must not** be `$defs`, `$ref`, `anyOf`, `required`, or `additionalProperties`; **no duplicate keys**. A name in `required` that `properties` does not declare is accepted by lite and pruned by `Canonical`. |
| `additionalProperties` | ✅ Value must be a **boolean** or an **object**; if omitted, **defaults to true**. |
| `patternProperties` | ❌ |
| `dependentSchemas` | ❌ |
@@ -134,9 +134,11 @@ This document summarizes how the **walle** validator handles each keyword in **J
| Topic | walle |
| --- | --- |
| Empty object subschema `{}` | **ANY** is expressed only when the **entire root** is `{}` or when **`additionalProperties`** is `{}`. A `{}` **inside** `properties` is **not** treated as ANY. |
-| `type` alongside `anyOf` / `$ref` | **Disallowed**; put `type` **inside** the `anyOf` branch or the `$ref` target. |
+| `type` alongside `anyOf` | **Allowed**—2020-12 applies it as a logical AND. lite accepts it and `Canonical` pushes it into every branch, dropping the branches it contradicts; if that leaves no branch, the subschema degrades to `{}`. |
+| `type` alongside `$ref` | **Allowed**—2020-12 applies it as a logical AND. lite accepts it and ultra folds it into the target. An empty intersection with the target's `type` admits no instance: lite still accepts it but `Canonical` degrades that subschema to `{}`, and strict and above reject it. |
| Keywords allowed on `object` | **`type`**, **`properties`**, **`required`**, **`additionalProperties`**, **`anyOf`**, **`$ref`**, plus annotations such as **`description`** / **`title`** where rules allow. |
-| Siblings of `anyOf` / `$ref` | Besides **`description`** / **`title`**, the **root** may also include **`$defs`** / **`$id`**. |
+| Siblings of `anyOf` | Constraint keywords are allowed, and the **root** may also include **`$defs`** / **`$id`**. `Canonical` distributes the constraints into every branch and leaves **`description`** / **`title`** where they are. |
+| Siblings of `$ref` | Constraint keywords are **allowed**. lite accepts them; ultra inlines the definition and keeps the stricter value for each shared keyword—see [validation-principles.md](./validation-principles.md). |
| Nesting and size | For example, **total `properties` keys across objects** and **nesting depth** **may be limited**—see **[walle.md](./walle.md)**. |
| Numeric and enum literals | Integers **decimal only**; floating-point **no scientific notation**; further bounds as in **walle.md**. |
diff --git a/docs/mfjs-walle-vs-draft-2020-12.zh.md b/docs/mfjs-walle-vs-draft-2020-12.zh.md
index 57a7c11..da2ddc5 100644
--- a/docs/mfjs-walle-vs-draft-2020-12.zh.md
+++ b/docs/mfjs-walle-vs-draft-2020-12.zh.md
@@ -36,13 +36,13 @@
| Keyword | walle |
| --- | --- |
| `allOf` | ❌ |
-| `anyOf` | ✅ 分支数量可能有限制;与 `type` / `$ref` **不得同级**(`type` 须在分支内) |
+| `anyOf` | ✅ 分支数量可能有限制;同级可带 `type` 等约束(按 AND 生效),`Canonical` 会把它们分发进每个分支 |
| `oneOf` | ❌ |
| `if` | ❌ |
| `then` | ❌ |
| `else` | ❌ |
| `not` | ❌ |
-| `properties` | ✅ `type: object` 时;**key 不可** 为 `$defs`、`$ref`、`anyOf`、`required`、`additionalProperties`;**不可重复**;`required` 中每项须在 `properties` 中声明 |
+| `properties` | ✅ `type: object` 时;**key 不可** 为 `$defs`、`$ref`、`anyOf`、`required`、`additionalProperties`;**不可重复**;`required` 中未在 `properties` 声明的项 lite 放行,`Canonical` 会把它剔除 |
| `additionalProperties` | ✅ 值为 **boolean 或 object**;未指定时 **默认 true** |
| `patternProperties` | ❌ |
| `dependentSchemas` | ❌ |
@@ -134,9 +134,11 @@
| 主题 | walle |
| --- | --- |
| 空 object subschema `{}` | 仅 **整份 root 为 `{}`** 或 **`additionalProperties` 值为 `{}`** 表示 **ANY**;**`properties` 内 `{}` 不自动视为 ANY** |
-| `type` 与 `anyOf` / `$ref` 同级 | **禁止**;`type` 须在 `anyOf` / `$ref` 目标内部 |
+| `type` 与 `anyOf` 同级 | **允许**(2020-12 按 AND 生效);lite 放行,`Canonical` 把 `type` 分发进每个分支,与父层矛盾的分支丢弃、全部丢弃则该子 schema 退化为 `{}` |
+| `type` 与 `$ref` 同级 | **允许**(2020-12 按 AND 生效);lite 放行,ultra 折叠进引用目标;与目标 `type` 交集为空时属恒假,lite 仍放行但 Canonical 把该子 schema 退化为 `{}`,strict 及以上拒绝 |
| `object` 上允许的 keyword | **仅** `type`、`properties`、`required`、`additionalProperties`、`anyOf`、`$ref`(及注解规则中的 `description` / `title` 等) |
-| `anyOf` / `$ref` 同级其它 keyword | 除 `description` / `title` 外,**root** 可额外有 `$defs` / `$id` |
+| `anyOf` 同级其它 keyword | 允许约束关键字,**root** 可额外有 `$defs` / `$id`;`Canonical` 把约束分发进各分支,`description` / `title` 留在原处 |
+| `$ref` 同级其它 keyword | **允许**约束关键字;lite 放行,ultra 内联展开并对同名关键字取更严的一侧,详见 [validation-principles.zh.md](./validation-principles.zh.md) |
| 嵌套与规模 | 如 **全 schema 中 object properties 数量可能有限制(累计)**、**嵌套层数可能有限制**(以 [walle.zh.md](./walle.zh.md) 为准) |
| 数值与枚举字面量 | 整数 **十进制**;浮点 **无科学计数法**;等 |
diff --git a/docs/validation-principles.md b/docs/validation-principles.md
new file mode 100644
index 0000000..9887049
--- /dev/null
+++ b/docs/validation-principles.md
@@ -0,0 +1,247 @@
+> 🌐 **Language:** English | [简体中文 (Chinese)](./validation-principles.zh.md)
+
+# walle Validation Rules and Canonicalization
+
+This document answers three questions: **why was my schema rejected**, **how do I fix it**, and **what will it be rewritten into once it passes**.
+
+For the full rule table see [walle.md](./walle.md), for the spec itself see [mfjs-spec.md](./mfjs-spec.md), and for a keyword-by-keyword comparison with JSON Schema 2020-12 see [mfjs-walle-vs-draft-2020-12.md](./mfjs-walle-vs-draft-2020-12.md).
+
+## 1. The four levels
+
+| Level | Behaviour |
+| --- | --- |
+| `loose` | No validation at all; every schema passes |
+| `lite` | Rejects misspelled schemas, unresolvable references, and non-terminating recursion |
+| `strict` | On top of `lite`, additionally rejects schemas with mistyped or self-contradictory numeric bounds |
+| `ultra` | The strictest; `Canonical` uses it to find everything that needs rewriting |
+
+`lite` cares about exactly one thing: **is this schema well-formed and satisfiable**. Constructs that are legal but unusable downstream are all accepted and left for `Canonical` to rewrite.
+
+**Rejected by the interface?** See section 2. **Accepted but the generated output is not what you expected?** See sections 3-5.
+
+## 2. Constructs that get rejected
+
+Everything below is rejected by `lite` (the interface default).
+
+### 2.1 A keyword's value is written incorrectly
+
+| Case | Error |
+| --- | --- |
+| `type` is not a string or an array of strings | `type must be string or array of strings` |
+| `type` array is empty | `type array cannot be empty` |
+| `type` array contains a non-string entry | `invalid type in type array` |
+| `type` is not one of the 7 types | `invalid type` |
+| `properties` is not an object | `properties must be an object` |
+| A property's schema is not an object | `property schema for 'x' must be an object` |
+| `required` is not an array | `required must be an array` |
+| `required` contains a non-string entry | `items in required array must be strings` |
+| `enum` is not an array | `enum must be an array` |
+| `enum` is an empty array | `enum array cannot be empty` |
+| `items` is not an object | `items must be an object` |
+| `anyOf` is not an array | `anyOf must be an array` |
+| `anyOf` is an empty array | `anyOf must have 1-500 items` |
+| `$ref` is not a string | `$ref must be a string` |
+| `pattern` is not a string | `pattern must be a string` |
+| `description` is not a string | `description must be a string` |
+| `additionalProperties` is not a boolean or an object | `additionalProperties must be a boolean or an object` |
+| `$defs` is not an object | `$defs must be an object` |
+| `$id` is not a string | `$id must be a string` |
+
+Size limits also apply at `lite`: 120000 bytes per schema, 30 levels of nesting, 3000 property keys across all objects, 1000 items per `enum`, and 500 `anyOf` branches.
+
+When a numeric bound keyword (`minLength` `maxLength` `minimum` `maximum` `minItems` `maxItems`) has a non-integer or negative value, `lite` accepts it and the rewrite corrects it; `strict` and above reject it.
+
+### 2.2 References that cannot be resolved
+
+Only references pointing at `$defs` inside this document are supported.
+
+| Case | Error |
+| --- | --- |
+| External, cross-file, or URL reference, e.g. `"$ref":"https://example.com/s.json"` | `references must start with #/$defs/` |
+| Points at a path outside `#/$defs/` | `references must start with #/$defs/` |
+| Empty definition name, i.e. `"$ref":"#/$defs/"` | `definition name cannot be empty` |
+| The referenced definition does not exist | `invalid $ref path: ...` |
+| Uses `$ref` but the schema has no `$defs` at all | `$defs not found for reference: ...` |
+| A `$defs` key contains `/` | `$defs property name 'a/b' cannot contain '/' character` |
+
+### 2.3 Nothing can ever satisfy it
+
+Only non-terminating recursion falls into this category. Contradictory `type` / `enum` / numeric bounds do not — they are equally unsatisfiable, but `lite` accepts them and the rewrite drops the contradicting constraint; see section 5.
+
+| Case | Error |
+| --- | --- |
+| Required properties form a reference cycle | `detected infinite recursion without termination condition` |
+| Mutual references where every hop is required | Same as above |
+| Array self-reference with `minItems >= 1` | Same as above |
+
+### 2.4 The reference graph is too complex
+
+Diamond reference chains (every definition referenced by several properties, each level pointing at the next definition) make reference traversal grow exponentially with the chain length. Once the traversal exceeds the built-in step budget (500000 steps) the schema is rejected:
+
+| Case | Error |
+| --- | --- |
+| Diamond / long-cycle reference chains exceeding the traversal step budget | `reference graph is too complex to validate within the step budget` |
+
+Honest schemas stay orders of magnitude below the budget; only deliberately crafted ones hit this.
+
+## 3. How keywords beside `$ref` are merged
+
+| Keyword | Merge rule |
+| --- | --- |
+| `minLength` `minItems` `minimum` | Keep the larger value |
+| `maxLength` `maxItems` `maximum` | Keep the smaller value |
+| `type` | Intersect; the intersection of `integer` and `number` is `integer`; an empty intersection is covered in section 5 |
+| `enum` | Intersect; an empty intersection is covered in section 5 |
+| `required` | Union |
+| `properties` | Union; a property described on both sides is merged one level deeper by the rules above |
+| `description` `title` `pattern` and all other keywords | The use site's value wins |
+
+The use site asks for `minLength: 5` while the definition says `1`; the merged result keeps `5`, and the definition's `maxLength: 50` comes along:
+
+```json
+{
+ "type": "object",
+ "properties": { "name": { "minLength": 5, "$ref": "#/$defs/Name" } },
+ "$defs": { "Name": { "type": "string", "minLength": 1, "maxLength": 50 } }
+}
+```
+
+Rewritten as:
+
+```json
+{
+ "type": "object",
+ "properties": {
+ "name": { "type": "string", "minLength": 5, "maxLength": 50 }
+ }
+}
+```
+
+The remaining cases (the table omits the outer `"type":"object"` and `"properties"` wrapper):
+
+| Use site + definition | Rewrite result | Notes |
+| --- | --- | --- |
+| `{"minLength":1,"$ref":"#/$defs/S"}`
`S = {"type":"string","minLength":1}` | `{"$ref":"#/$defs/S"}` with `S` unchanged | Both sides agree, so the duplicate is dropped and the definition stays shared |
+| `{"maxLength":20,"$ref":"#/$defs/S"}`
`S = {"type":"string","minLength":1}` | `{"type":"string","minLength":1,"maxLength":20}` | The keyword is absent from the definition, so it is folded in directly |
+| `{"description":"user name","$ref":"#/$defs/S"}` | Kept as written; passes at every level | Annotations do not affect constraints |
+| `short: {"maxLength":10,"$ref":"#/$defs/S"}`
`long: {"maxLength":99,"$ref":"#/$defs/S"}`
`S = {"type":"string","minLength":1}` | `short: {"type":"string","minLength":1,"maxLength":10}`
`long: {"type":"string","minLength":1,"maxLength":99}` | Each use site specialises independently; they do not affect one another |
+
+**Exception: when the referenced definition is recursive, sibling constraints are dropped.**
+
+```json
+{
+ "type": "object",
+ "properties": { "tree": { "minLength": 3, "$ref": "#/$defs/N" } },
+ "$defs": {
+ "N": { "type": "object", "properties": { "child": { "$ref": "#/$defs/N" } } }
+ }
+}
+```
+
+After the rewrite `minLength: 3` is gone and `tree` is constrained by `N` alone. If you need that constraint to take effect, do not write it at the reference site of a recursive definition; move it into the definition itself.
+
+## 4. How keywords beside `anyOf` are distributed
+
+A constraint written beside `anyOf` applies together with whichever branch matches. The rewrite **pushes the parent constraint into every branch**:
+
+```json
+{
+ "type": "object",
+ "properties": {
+ "v": { "type": "string", "anyOf": [{ "minLength": 1 }, { "maxLength": 9 }] }
+ }
+}
+```
+
+Rewritten as:
+
+```json
+{
+ "type": "object",
+ "properties": {
+ "v": {
+ "anyOf": [
+ { "type": "string", "minLength": 1 },
+ { "type": "string", "maxLength": 9 }
+ ]
+ }
+ }
+}
+```
+
+Distribution rules:
+
+- **Same-name keywords merge by the stricter-wins rules of section 3**: a parent `minLength: 20` meeting a branch's `minLength: 10` leaves the branch with 20.
+- **A branch that contradicts the parent is dropped**, because no instance can ever reach it. A parent `"type":"string"` over `[{"minLength":1},{"type":"integer"}]` makes the `integer` branch disappear.
+- **When every branch is dropped the field is emptied to `{}`** — such a schema had no satisfying instance to begin with.
+- **`description` / `title` are not distributed**; they stay at the parent level because they are not constraints.
+
+Contradictions count when a branch uses `$ref` too: a parent `"type":"string"` over a branch referencing a definition of `{"type":"number"}` gets that branch dropped as well.
+
+## 5. Other constructs that get rewritten
+
+`lite` accepts all of these; `Canonical` rewrites them. The table omits the outer wrapper.
+
+| Case | Input | Rewrite result |
+| --- | --- | --- |
+| Uses an unsupported keyword | `{"type":"string","format":"uuid"}` | `{"type":"string"}` — only that keyword is removed |
+| `$defs` / `$id` not at the root | `$defs` inside a subschema | The keyword is removed, everything else kept |
+| Duplicates in a `type` array or in `required` | `{"type":["string","string"]}` | `{"type":["string"]}` |
+| Negative bound value | `{"type":"string","minLength":-1}` | `{"type":"string","minLength":0}` |
+| `enum` holds values the `type` rules out | `{"type":"string","enum":["a",1,"b"]}` | `{"type":"string","enum":["a","b"]}` — only the ill-typed values go |
+| Every `enum` value is ruled out by `type` | `{"type":["null"],"enum":[false]}` | `{"type":["null"]}` — the whole `enum` disappears |
+| Lower bound above upper bound | `{"type":"string","minLength":10,"maxLength":2}` | `{}` — the field loses all constraints |
+| Multiple `type`s plus other structural keywords | `{"type":["string","integer"],"minLength":1}` | `{}` — the field loses all constraints |
+| `enum` has no overlap with the definition's `enum` | `{"type":"string","enum":["x"],"$ref":"#/$defs/S"}`
`S = {"type":"string","enum":["y"],"minLength":9}` | `{"type":"string","minLength":9}` — only the `enum` is dropped |
+| `type` has no overlap with the definition's `type` | `{"type":"string","minLength":3,"$ref":"#/$defs/S"}`
`S = {"type":"number","minimum":5}` | `{}` — the field loses all constraints |
+
+`required` is listed separately; these constructs are all legal under the spec, and the handling rules are:
+
+| Case | Input | Rewrite result |
+| --- | --- | --- |
+| A required property is not declared in `properties` | `properties: {"a":{...}}`
`required: ["a","b"]` | `required: ["a"]` — only `b` is removed; `a` stays required |
+| `required` without `properties` | `{"type":"object","required":["a"]}` | `{"type":"object"}` — the whole `required` disappears |
+| `required` with a non-`object` `type` | `{"type":"string","required":["a"]}` | `{"type":"string"}` — `required` only applies to objects anyway |
+
+The empty string is a perfectly normal property name: the spec puts no constraint on the keys of `properties`, and the downstream constraint engine can generate `{"": ...}`, so `{"":{"type":"string"}}` is kept as written and `required: [""]` works as usual. It follows the same rule as any other name: it is pruned only when it is not declared in `properties`.
+
+The last four rows of the first table (lower bound above upper bound, multiple `type`s, `enum` / `type` with no overlap on either side) are rejected outright at `strict` and above, because no instance can satisfy them. Every other row is accepted at all levels.
+
+The rewrite's trade-off is **drop only the one contradicting keyword and keep whatever the two sides still agree on**. In the row where the `enum`s do not overlap, both sides agree the value is a string — they just disagree on which strings — so `type` stays, `minLength` keeps the stricter 9, and only the `enum` disappears.
+
+A contradiction on `type` is the exception: the whole field is emptied. Every other keyword only means something relative to a type, and combining one side's string length with the other side's numeric lower bound on a typeless field describes something that does not exist.
+
+A full clash between `enum` and a sibling `type` is a compromise: one of the two has to go, and `type` stays. Keeping the `enum` would admit values of a type the schema explicitly ruled out; keeping the `type` only loosens, and is the tightest result still expressible.
+
+## 6. Can recursive references be used
+
+Self-reference itself is supported. The criterion is **whether a finite JSON document can be constructed**, not whether self-reference exists.
+
+| Construct | Verdict | Reason |
+| --- | --- | --- |
+| `{"properties":{"next":{"$ref":"#"}}}` | Passes | `next` is optional; `{}` satisfies it |
+| `{"properties":{"children":{"type":"array","items":{"$ref":"#"}}},"required":["children"]}` | Passes | `{"children":[]}` satisfies it; an empty array asks nothing of `items` |
+| `{"properties":{"next":{"$ref":"#"}},"required":["next"]}` | Rejected | Every level demands a next one |
+| The array version of the previous row, plus `"minItems":1` | Rejected | The array cannot be empty, so the recursion cannot bottom out |
+
+## 7. Unsupported keywords
+
+`allOf`, `oneOf`, `not`, `if` / `then` / `else`, `const`, `format`, `$schema`, `$comment`, `$anchor`, `$dynamicRef`, and any other unknown keyword: `lite` accepts them, `Canonical` deletes them.
+
+## 8. Behaviour changes versus the old version
+
+| Construct | Old behaviour | Now |
+| --- | --- | --- |
+| Constraints such as `type` or `minLength` beside `$ref` | Rejected | Accepted; merged stricter-wins on rewrite |
+| `type` beside `$ref` with compatible types | Rejected | Accepted |
+| `type` beside `anyOf` | Rejected | Accepted; distributed into every branch on rewrite |
+| Same keyword both at the parent and inside an `anyOf` branch | Rejected | Accepted; distributed and merged stricter-wins on rewrite |
+| A required property not declared in `properties` | Rejected | Accepted; only that entry is removed on rewrite |
+| `required` without `properties`, or with a non-object `type` | Rejected | Accepted; `required` is removed on rewrite |
+| `enum` holding values the `type` rules out | Rejected | Accepted; those values are removed on rewrite |
+| Empty-string property name | The property was deleted on rewrite | Kept as written; listing it in `required` works as usual |
+| Required properties forming a reference cycle | Accepted, but no valid output could be generated | Rejected |
+| Required array self-reference without `minItems` | Rejected (a misjudgement) | Accepted |
+| Lower bound above upper bound | Accepted; both bounds were deleted on rewrite | `lite` accepts but degrades the field to `{}`; `strict` and above reject |
+
+There is one more pervasive change in rewrite results: previously, a single spot needing a rewrite could make `Canonical` return an empty `{}` and lose the whole set of constraints; now only the field at fault degrades, and everything else is preserved intact.
diff --git a/docs/validation-principles.zh.md b/docs/validation-principles.zh.md
new file mode 100644
index 0000000..ea8e5bd
--- /dev/null
+++ b/docs/validation-principles.zh.md
@@ -0,0 +1,247 @@
+> 🌐 **语言 (Language):** [English](./validation-principles.md) | 简体中文
+
+# walle 校验规则与规范化说明
+
+这份文档回答三个问题:**我的 schema 为什么被拒了**、**怎么改**、**通过之后它会被改写成什么样**。
+
+规则总表见 [walle.zh.md](./walle.zh.md),规范定义见 [mfjs-spec.zh.md](./mfjs-spec.zh.md),与 JSON Schema 2020-12 的逐条对照见 [mfjs-walle-vs-draft-2020-12.zh.md](./mfjs-walle-vs-draft-2020-12.zh.md)。
+
+## 一、四个级别
+
+| 级别 | 行为 |
+| --- | --- |
+| `loose` | 完全不校验,任何 schema 都通过 |
+| `lite` | 拒绝写错的、引用解析不了的、以及无终止递归 |
+| `strict` | 在 `lite` 之上,额外拒绝数值边界写错或自相矛盾的 schema |
+| `ultra` | 最严,`Canonical` 用它找出所有需要改写的地方 |
+
+`lite` 只管一件事:**这份 schema 是不是合法且有解**。合法但下游用不了的写法它一律放行,交给 `Canonical` 改写。
+
+**被接口拒了**看第二节;**通过了但生成结果不符合预期**看第三到五节。
+
+## 二、会被拒绝的写法
+
+以下都是 `lite`(接口默认)会拒绝的
+
+### 2.1 关键字的值写错了
+
+| 情形 | 报错 |
+| --- | --- |
+| `type` 不是字符串或字符串数组 | `type must be string or array of strings` |
+| `type` 数组为空 | `type array cannot be empty` |
+| `type` 数组里有非字符串元素 | `invalid type in type array` |
+| `type` 不是那 7 种之一 | `invalid type` |
+| `properties` 不是对象 | `properties must be an object` |
+| 某个属性的 schema 不是对象 | `property schema for 'x' must be an object` |
+| `required` 不是数组 | `required must be an array` |
+| `required` 里有非字符串元素 | `items in required array must be strings` |
+| `enum` 不是数组 | `enum must be an array` |
+| `enum` 是空数组 | `enum array cannot be empty` |
+| `items` 不是对象 | `items must be an object` |
+| `anyOf` 不是数组 | `anyOf must be an array` |
+| `anyOf` 是空数组 | `anyOf must have 1-500 items` |
+| `$ref` 不是字符串 | `$ref must be a string` |
+| `pattern` 不是字符串 | `pattern must be a string` |
+| `description` 不是字符串 | `description must be a string` |
+| `additionalProperties` 不是布尔值或对象 | `additionalProperties must be a boolean or an object` |
+| `$defs` 不是对象 | `$defs must be an object` |
+| `$id` 不是字符串 | `$id must be a string` |
+
+规模上限同样在 `lite` 生效:整份 schema 120000 字节、嵌套 30 层、所有对象累计 3000 个属性键、单个 `enum` 1000 项、`anyOf` 500 个分支。
+
+数值边界关键字(`minLength` `maxLength` `minimum` `maximum` `minItems` `maxItems`)的值不是整数或为负数时,`lite` 放行并在改写时纠正,`strict` 及以上才拒绝。
+
+### 2.2 引用无法解析
+
+只支持指向本文档内部 `$defs` 的引用。
+
+| 情形 | 报错 |
+| --- | --- |
+| 外部、跨文件、URL 引用,如 `"$ref":"https://example.com/s.json"` | `references must start with #/$defs/` |
+| 指向 `#/$defs/` 以外的路径 | `references must start with #/$defs/` |
+| 定义名为空,即 `"$ref":"#/$defs/"` | `definition name cannot be empty` |
+| 引用的定义不存在 | `invalid $ref path: ...` |
+| 用了 `$ref` 但整份 schema 没有 `$defs` | `$defs not found for reference: ...` |
+| `$defs` 的键名里有 `/` | `$defs property name 'a/b' cannot contain '/' character` |
+
+### 2.3 没有任何数据能满足
+
+只有无终止递归属于这一类。矛盾的 `type` / `enum` / 数值边界不在此列——它们同样无解,但 `lite` 放行、改写时丢掉矛盾的约束,详见第五节。
+
+| 情形 | 报错 |
+| --- | --- |
+| 必填属性构成引用环 | `detected infinite recursion without termination condition` |
+| 互相引用且每一跳都必填 | 同上 |
+| 数组自引用且 `minItems >= 1` | 同上 |
+
+### 2.4 引用图过于复杂
+
+菱形引用链(每个定义被多个属性引用、再层层指向下一个定义)会让引用遍历的成本随链长指数增长。遍历步数超过内置预算(500000 步)时直接拒绝:
+
+| 情形 | 报错 |
+| --- | --- |
+| 菱形/长环引用链,引用遍历超出步数预算 | `reference graph is too complex to validate within the step budget` |
+
+正常 schema 的遍历量比这低几个数量级,只有刻意构造的 schema 才会碰到这条。
+
+## 三、$ref 同级关键字会被怎么合并
+
+| 关键字 | 合并方式 |
+| --- | --- |
+| `minLength` `minItems` `minimum` | 取较大值 |
+| `maxLength` `maxItems` `maximum` | 取较小值 |
+| `type` | 取交集,`integer` 与 `number` 的交集是 `integer`;交集为空见第五节 |
+| `enum` | 取交集;交集为空见第五节 |
+| `required` | 取并集 |
+| `properties` | 取并集;两侧都描述的属性按上面这些规则再合并一层 |
+| `description` `title` `pattern` 等其余关键字 | 以使用处的值为准 |
+
+使用处要求 `minLength: 5`、定义里是 `1`,合并结果取 `5`,定义里的 `maxLength: 50` 一并带过来:
+
+```json
+{
+ "type": "object",
+ "properties": { "name": { "minLength": 5, "$ref": "#/$defs/Name" } },
+ "$defs": { "Name": { "type": "string", "minLength": 1, "maxLength": 50 } }
+}
+```
+
+改写为:
+
+```json
+{
+ "type": "object",
+ "properties": {
+ "name": { "type": "string", "minLength": 5, "maxLength": 50 }
+ }
+}
+```
+
+其余几种情形(下表省略了外层的 `"type":"object"` 和 `"properties"` 包裹):
+
+| 使用处 + 定义 | 改写结果 | 说明 |
+| --- | --- | --- |
+| `{"minLength":1,"$ref":"#/$defs/S"}`
`S = {"type":"string","minLength":1}` | `{"$ref":"#/$defs/S"}`,`S` 不变 | 两侧值相同,删掉重复的那个、继续共享定义 |
+| `{"maxLength":20,"$ref":"#/$defs/S"}`
`S = {"type":"string","minLength":1}` | `{"type":"string","minLength":1,"maxLength":20}` | 使用处的关键字定义里没有,直接并入 |
+| `{"description":"用户名","$ref":"#/$defs/S"}` | 原样保留,所有级别都通过 | 注解不影响约束 |
+| `short: {"maxLength":10,"$ref":"#/$defs/S"}`
`long: {"maxLength":99,"$ref":"#/$defs/S"}`
`S = {"type":"string","minLength":1}` | `short: {"type":"string","minLength":1,"maxLength":10}`
`long: {"type":"string","minLength":1,"maxLength":99}` | 多个使用处各自独立特化,互不影响 |
+
+**例外:被引用的定义是递归的,同级约束会被丢弃。**
+
+```json
+{
+ "type": "object",
+ "properties": { "tree": { "minLength": 3, "$ref": "#/$defs/N" } },
+ "$defs": {
+ "N": { "type": "object", "properties": { "child": { "$ref": "#/$defs/N" } } }
+ }
+}
+```
+
+改写后 `minLength: 3` 消失,`tree` 只受 `N` 约束。需要这个约束生效的话,不要写在递归定义的引用处,改为写进定义本身。
+
+## 四、anyOf 同级关键字会被怎么分发
+
+写在 `anyOf` 旁边的约束,和命中的那个分支同时生效。改写时把父层的约束**逐个分发进每个分支**:
+
+```json
+{
+ "type": "object",
+ "properties": {
+ "v": { "type": "string", "anyOf": [{ "minLength": 1 }, { "maxLength": 9 }] }
+ }
+}
+```
+
+改写为:
+
+```json
+{
+ "type": "object",
+ "properties": {
+ "v": {
+ "anyOf": [
+ { "type": "string", "minLength": 1 },
+ { "type": "string", "maxLength": 9 }
+ ]
+ }
+ }
+}
+```
+
+分发规则:
+
+- **同名关键字按第三节的取严规则合并**,父层 `minLength: 20` 遇上分支的 `minLength: 10`,分支留下 20。
+- **和父层矛盾的分支直接删掉**,因为没有数据能走到它。父层 `"type":"string"` 配上 `[{"minLength":1},{"type":"integer"}]`,`integer` 那个分支消失。
+- **所有分支都被删掉时该字段清空为 `{}`**,这种 schema 本来就没有数据能满足。
+- **`description` / `title` 不分发**,留在父层,它们不是约束。
+
+分支里写 `$ref` 时,矛盾也算在内:父层 `"type":"string"` 而分支引用的定义是 `{"type":"number"}`,该分支同样被删。
+
+## 五、其他会被改写的写法
+
+这些 `lite` 都放行,`Canonical` 会改写。下表省略了外层包裹。
+
+| 情形 | 输入 | 改写结果 |
+| --- | --- | --- |
+| 用了不支持的关键字 | `{"type":"string","format":"uuid"}` | `{"type":"string"}`,只删该关键字 |
+| `$defs` / `$id` 没写在根层 | 子 schema 里带 `$defs` | 删掉该关键字,其余保留 |
+| `type` 数组或 `required` 里有重复项 | `{"type":["string","string"]}` | `{"type":["string"]}` |
+| 边界值为负 | `{"type":"string","minLength":-1}` | `{"type":"string","minLength":0}` |
+| `enum` 里有 `type` 不允许的值 | `{"type":"string","enum":["a",1,"b"]}` | `{"type":"string","enum":["a","b"]}`,只去掉不合类型的值 |
+| `enum` 里全部值都不合 `type` | `{"type":["null"],"enum":[false]}` | `{"type":["null"]}`,`enum` 整个消失 |
+| 下界大于上界 | `{"type":"string","minLength":10,"maxLength":2}` | `{}`,该字段失去全部约束 |
+| 多个 `type` 且带其他结构关键字 | `{"type":["string","integer"],"minLength":1}` | `{}`,该字段失去全部约束 |
+| `enum` 与定义的 `enum` 无交集 | `{"type":"string","enum":["x"],"$ref":"#/$defs/S"}`
`S = {"type":"string","enum":["y"],"minLength":9}` | `{"type":"string","minLength":9}`,只丢掉 `enum` |
+| `type` 与定义的 `type` 无交集 | `{"type":"string","minLength":3,"$ref":"#/$defs/S"}`
`S = {"type":"number","minimum":5}` | `{}`,该字段失去全部约束 |
+
+`required` 单独列出来,这几种写法在规范里都合法,处理规则:
+
+| 情形 | 输入 | 改写结果 |
+| --- | --- | --- |
+| 必填的属性没在 `properties` 里声明 | `properties: {"a":{...}}`
`required: ["a","b"]` | `required: ["a"]`,只去掉 `b`,`a` 仍然必填 |
+| 有 `required` 但没有 `properties` | `{"type":"object","required":["a"]}` | `{"type":"object"}`,`required` 整个消失 |
+| 有 `required` 但 `type` 不是 `object` | `{"type":"string","required":["a"]}` | `{"type":"string"}`,`required` 本来就只对对象生效 |
+
+空字符串是个正常的属性名,规范没有限制 `properties` 的键,下游的约束引擎也能生成 `{"": ...}`,所以 `{"":{"type":"string"}}` 原样保留、`required: [""]` 也照常生效。它和别的名字走同一条规则:只有没在 `properties` 里声明时才会被剔除。
+
+第一张表的最后四行(下界大于上界、多个 `type`、两处 `enum` / `type` 无交集)`strict` 及以上级别会直接拒绝,因为它们都没有任何数据能满足。其余各行所有级别都放行。
+
+改写的取舍是**只丢掉矛盾的那一个关键字,两侧还能达成一致的部分留着**。`enum` 与定义的 `enum` 无交集那行,两侧都同意值是字符串,只是不同意是哪些字符串,所以 `type` 保留、`minLength` 按取严规则留下更大的 9,只有 `enum` 消失。
+
+矛盾出在 `type` 上是例外,整个字段清空:其余关键字都要依附于某个类型才有意义,把一侧的字符串长度和另一侧的数值下限凑在一个无类型的字段里,描述的是一个不存在的东西。
+
+`enum` 和同级 `type` 全冲突时是个折中:两者留一个,留下的是 `type`。留 `enum` 会放进 schema 明确排除掉的类型的值,留 `type` 只是放宽,是能表达出来的最紧的结果。
+
+## 六、递归引用能不能用
+
+自引用本身是支持的。判据是**能否构造出有限的 JSON**,而不是有没有自引用。
+
+| 写法 | 判定 | 原因 |
+| --- | --- | --- |
+| `{"properties":{"next":{"$ref":"#"}}}` | 通过 | `next` 可选,`{}` 就满足 |
+| `{"properties":{"children":{"type":"array","items":{"$ref":"#"}}},"required":["children"]}` | 通过 | `{"children":[]}` 满足,空数组对 `items` 无要求 |
+| `{"properties":{"next":{"$ref":"#"}},"required":["next"]}` | 拒绝 | 每一层都还需要下一层 |
+| 上一行的数组版本再加 `"minItems":1` | 拒绝 | 数组不能为空,递归无法收尾 |
+
+## 七、不支持的关键字
+
+`allOf`、`oneOf`、`not`、`if` / `then` / `else`、`const`、`format`、`$schema`、`$comment`、`$anchor`、`$dynamicRef`,以及任何其他未知关键字:`lite` 放行,`Canonical` 删除。
+
+## 八、与旧版本的行为差异
+
+| 写法 | 旧行为 | 现在 |
+| --- | --- | --- |
+| `$ref` 同级带 `type`、`minLength` 等约束 | 拒绝 | 放行,改写时合并取严 |
+| `type` 与 `$ref` 同级且类型兼容 | 拒绝 | 放行 |
+| `type` 与 `anyOf` 同级 | 拒绝 | 放行,改写时分发进各分支 |
+| 同一关键字既在父层又在 `anyOf` 分支里 | 拒绝 | 放行,改写时分发并取严 |
+| 必填的属性没在 `properties` 里声明 | 拒绝 | 放行,改写时只去掉这一项 |
+| 有 `required` 但没有 `properties`、或 `type` 不是对象 | 拒绝 | 放行,改写时删掉 `required` |
+| `enum` 里有 `type` 不允许的值 | 拒绝 | 放行,改写时去掉这些值 |
+| 属性名是空字符串 | 改写时删掉该属性 | 原样保留,`required` 里列它也照常生效 |
+| 必填属性构成引用环 | 放行,但无法生成有效结果 | 拒绝 |
+| 必填数组自引用、未写 `minItems` | 拒绝(误判) | 放行 |
+| 下界大于上界 | 放行,改写时删掉两个边界 | `lite` 放行但该字段退化为 `{}`,`strict` 及以上拒绝 |
+
+改写结果还有一处普遍变化:过去只要 schema 里有一处需要改写,`Canonical` 就可能返回空的 `{}`、丢掉整份约束;现在只有出问题的那个字段会退化,其余部分完整保留。
diff --git a/docs/walle.md b/docs/walle.md
index 34faa91..53dec7f 100644
--- a/docs/walle.md
+++ b/docs/walle.md
@@ -17,24 +17,24 @@ This document describes how **walle** validates JSON Schemas and classifies erro
| Category | Rules | Details |
| --- | --- | --- |
-| Structural errors | Every subschema must declare `type` explicitly. If `anyOf` or `$ref` is present, `type` must appear **inside** that `anyOf` / `$ref`, not beside it at the same level. | Two exceptions:
Case 1: the entire schema is `{}` → ANY.
Case 2: `"additionalProperties": {}` → ANY.
Note: in all **other** cases, `{}` is **not** inferred as ANY—for example:
"properties": {
"key1": {},
"key2": {}
} |
+| Structural errors | Every subschema must declare `type` explicitly. Beside `anyOf` or `$ref` it is legal—2020-12 evaluates siblings as a logical AND—so lite accepts it: `Canonical` pushes it into each `anyOf` branch, or folds it into the referenced schema. An empty intersection with the referenced `type` admits no instance at all: lite still accepts it but `Canonical` degrades that subschema to `{}`, and strict and above reject it. | Two exceptions:"properties": {
"key1": {},
"key2": {}
} |
| | Only the seven types `null` / `boolean` / `object` / `array` / `number` / `integer` / `string` are supported; the root schema must be a JSON object. | |
| | Only keywords allowed by MFJS. | Two cases: illegal keywords, and keywords that are legal in JSON Schema but not supported by MFJS. |
| | Keyword placement must follow JSON Schema conventions. | For `type: object`, only keywords such as `type`, `properties`, `required`, `additionalProperties`, `anyOf`, and `$ref` apply (see MFJS for the full list). |
| | Object: every name in `required` must be declared in `properties`. | |
| | Object: `properties` keys must be unique. | |
-| | Object: nesting depth and property count are capped. | A schema may have up to **100** object properties in total, with up to **5** levels of nesting (OpenAI-style limit, for performance). |
+| | Object: nesting depth and property count are capped. | Defaults: **3000** `properties` keys across all objects, **30** levels of nesting, and **120000** bytes per schema. Callers can change these with `WithMaxTotalPropertiesKeysNum` / `WithMaxSchemaDepth` / `WithMaxSchemaSize`. |
| | Object: `properties` keys must not be named `"$defs"`, `"$ref"`, `"anyOf"`, `"required"`, or `"additionalProperties"`. | |
-| | The `type` keyword must not sit beside `anyOf` / `$ref`; `type` belongs **inside** them. | |
-| | `anyOf` must have between **1** and **10** items. | |
+| | The `type` keyword may sit beside `anyOf` or `$ref`, and lite accepts both. Ultra reports it so that `Canonical` pushes it into each `anyOf` branch, or folds it into the referenced schema. | See [validation-principles.md](./validation-principles.md) for the per-level behaviour. |
+| | `anyOf` must have between **1** and **500** items by default (`WithMaxAnyOfItems`). | |
| | `$defs` and `$id` may only appear at the **root**. | |
| | `$ref` must resolve within this schema or its `$defs`; **no** remote, cross-file, or URL refs. | Self-reference: `"$ref": "#"`. |
-| | `$ref` / `$defs` must admit a sound termination condition; **infinite** recursive loops are forbidden. | |
+| | `$ref` / `$defs` must admit a sound termination condition; **infinite** recursive loops are forbidden. | The test is whether a finite instance exists: a cycle through an optional property, or through an array without `minItems`, terminates; a cycle through a required property, or through an array with `minItems >= 1`, is rejected. |
| | `$ref` may only appear where allowed. | For example: `properties`, `$defs`, `additionalProperties`, `anyOf`, `items`, or the root. |
-| | Array: enum size limits. | A schema may have up to **500** enum values across all enum properties. For a single `number` / `integer` enum whose values are treated as strings, when there are more than **250** values the total string length (after number/integer → string) must not exceed **7,500** characters. |
+| | Array: enum size limits. | A single `enum` may hold up to **1000** values by default. A separate cap limits their combined string length to **75000** characters, but it is only checked once an `enum` exceeds **2500** values—above the 1000-value cap—so it never fires under the default config, only after a caller raises the item cap with `WithMaxEnumItems`. |
| | Array: `items` may be omitted; if present it must not be empty. | |
-| | Total string-size limits. | Same enum caps as above for string enums. Length is measured after Go’s `encoding/json` marshaling; **whitespace is not** counted toward that total. |
-| | Beside `anyOf` / `$ref`, only `description` and `title` are allowed at the same level; at the **root**, `$defs` and `$id` may also appear. | Strict rule to avoid type / keyword mismatches after expansion. |
+| | Total string-size limits. | As above: **1000** values per `enum` by default, a **75000**-character total with a **2500**-value trigger. Length is measured after Go’s `encoding/json` marshaling; **whitespace is not** counted toward that total. |
+| | Constraint keywords are allowed beside both `anyOf` and `$ref` (plus `$defs` / `$id` at the **root**). | Siblings are legal under 2020-12's AND semantics, so lite accepts them. Ultra reports them so that `Canonical` distributes the ones beside `anyOf` into every branch and inlines the ones beside `$ref`, keeping the stricter value for each shared keyword either way. |
| | `default` is only allowed for `boolean`, `number`, `string`, `integer`, and `null`, and must match the declared type. | |
| Data type errors | `type` must agree with `enum` values (e.g. not `integer` with `3.67`). | |
| | The value of `type` must be a string (or an MFJS-allowed `type` array). | |
@@ -47,5 +47,5 @@ This document describes how **walle** validates JSON Schemas and classifies erro
| | `description`, `title`, and `$id` must be strings. | |
| | `anyOf` must be an array; each item must be a valid subschema. | |
| | The value of `additionalPropertis` may only be `boolean` or `object`. | If omitted, the default is `true`. |
-| | Any `min*` / `max*` pair must satisfy `min <= max`. | With `"type": "integer"`, floating-point `minimum` / `maximum` are accepted at walle **Normal** validation level, but the enforcer **truncates toward zero**. |
+| | Any `min*` / `max*` pair must satisfy `min <= max`. | `min > max` cannot be satisfied by any instance: lite accepts it but `Canonical` degrades that subschema to `{}`, while strict and above reject it. Separately, with `"type": "integer"` floating-point `minimum` / `maximum` are accepted, but the enforcer **truncates toward zero**. |
| | `$defs` names must not contain `/`. | |
diff --git a/docs/walle.zh.md b/docs/walle.zh.md
index d6273aa..5a2e222 100644
--- a/docs/walle.zh.md
+++ b/docs/walle.zh.md
@@ -17,24 +17,24 @@
| Categories/错误分类 | rules | details |
| --- | --- | --- |
-| Structural Errors/结构错误 | subschema需要显示指定type字段, 如果是anyOf/$ref 那么type必须定义在anyOf/$ref内部,不能在anyOf/$ref同级目录 | 两种例外:"properties": {
"key1": {},
"key2": {}
} |
+| Structural Errors/结构错误 | subschema需要显式指定type字段。与 anyOf 或 $ref 同级时 type 合法(2020-12 下同级关键字按 AND 生效),lite 放行:与 anyOf 同级时 Canonical 把 type 分发进每个分支,与 $ref 同级时折叠进被引用的 schema。与被引用 schema 的 type 交集为空时属恒假,lite 仍放行但 Canonical 把该子 schema 退化为 {},strict 及以上拒绝 | 两种例外:"properties": {
"key1": {},
"key2": {}
} |
| | 只支持null/boolean/object/array/number/integer/string 这7种types, root schema必须是dict | |
| | 只支持MFJS约定的范围的keywords | 分为两种情况:非法的keyword、合法但是MFJS不支持的keyword |
| | 各种keywords的位置要符合json schema规范 | object类型中只允许type/properties/required/additionalProperties/anyOf/$ref 这些keywords |
| | Object: required 列举的字段必须是 properties 中声明过的 | |
| | Object: properties中的keys不能重复 | |
-| | Object: Objects have limitations on nesting depth and size | schema may have up to 100 object properties total, with up to 5 levels of nesting. |
+| | Object: Objects have limitations on nesting depth and size | 默认上限:所有 object 累计 3000 个 properties 键、嵌套 30 层、整份 schema 120000 字节。调用方可用 WithMaxTotalPropertiesKeysNum / WithMaxSchemaDepth / WithMaxSchemaSize 调整 |
| | Object: properties的key的名字不能是"$defs"/"$ref"/"anyOf"/"required"/"additionalProperties" | |
-| | type keyword不能与 anyOf/$ref 存在于同级目录,type应该位于anyOf/$ref的内部 | |
-| | anyOf 元素个数>=1, <= 10 | |
+| | type keyword 与 anyOf 或 $ref 同级都合法,lite 放行;ultra 报出来,由 Canonical 分发进 anyOf 各分支、或折叠进被引用的 schema | 分级行为详见 [validation-principles.zh.md](./validation-principles.zh.md) |
+| | anyOf 元素个数>=1, <= 500(默认,可用 WithMaxAnyOfItems 调整) | |
| | $defs/$id 只能定义在root level | |
| | $ref 需要指向本schema自身或者本schema的subschema 的合法$defs,不支持remote/跨文件/url | 指向自身: "$ref": "#" |
-| | $ref/$defs 需要有合理终止条件,禁止infinite recursive loop | |
+| | $ref/$defs 需要有合理终止条件,禁止infinite recursive loop | 判据是能否构造有限实例:可选属性成环、数组成环且未设 minItems 都可终止;必填属性成环、minItems >= 1 的数组成环则拒绝 |
| | $ref的位置要合法 | 可以存在于:"properties", "defs", "additionalProperties", "anyOf", "items", "root" 相关位置处 |
-| | Array: Limitations on enum size | schema may have up to 500 enum values across all enum properties. For a single/number/integer enum property with string values, the total string length(number/integer->string) of all enum values cannot exceed 7,500 characters when there are more than 250 enum values. |
+| | Array: Limitations on enum size | 单个 enum 默认最多 1000 项。另有一条字符串总长限制(上限 75000 字符),但它只在 enum 项数超过 2500 时才检查——该阈值高于 1000 的项数上限,因此默认配置下不会触发,只有调用方用 WithMaxEnumItems 放宽项数后才生效 |
| | Array: "items"可以不定义,如果定义则内容不能为空 | |
-| | Limitations on total string size | schema may have up to 500 enum values across all enum properties. For a single enum property with string values, the total string length of all enum values cannot exceed 7,500 characters when there are more than 250 enum values.是基于go 标准库json.Marshal统计,因此空白字符不被计入字符串大小 |
-| | anyOf/$ref同级目录下只能有$description/$title关键字,如果是root则可以有"$defs"/"$id"关键字 | 强限制,避免展开之后的各种复杂类型/关键字不匹配情况出现 |
+| | Limitations on total string size | 同上:enum 项数默认上限 1000,字符串总长上限 75000 字符、触发阈值 2500 项。长度基于 go 标准库 json.Marshal 统计,因此空白字符不被计入字符串大小 |
+| | anyOf 与 $ref 同级都允许约束关键字,root 额外允许"$defs"/"$id" | 按 2020-12 的 AND 语义合法:lite 放行。ultra 报出来,由 Canonical 把 anyOf 同级的约束分发进每个分支、把 $ref 同级的内联展开,同名关键字都取更严的一侧 |
| | default关键字只支持boolean/number/string/integer/null这些类型,默认值需要与类型匹配 | |
| Data Type Errors/类型错误 | type和enum value类型不匹配,如integer vs 3.67 | |
| | type的value类型必须是字符串 | |
@@ -47,5 +47,5 @@
| | description/title/$id 类型必须是字符串 | |
| | anyOf 类型必须是数组,每个items必须是合法的subSchema | |
| | additionalPropertis value的类型只能是boolean或者object | 如果不指定,默认值是true |
-| | 各类型存在的min/max需要满足 min <= max 的限制条件 | "type": "integer",但是mininum/maxinum的数值是浮点,walle validation level 为Normal时schema视为合法,但enforcer会进行Rounding,算法选择Rounding toward to zero (Truncate) |
+| | 各类型存在的min/max需要满足 min <= max 的限制条件 | min > max 属恒假:lite 放行但 Canonical 把该子 schema 退化为 {},strict 及以上拒绝。另:"type": "integer" 而 mininum/maxinum 为浮点时 schema 视为合法,但 enforcer 会进行 Rounding,算法选择 Rounding toward to zero (Truncate) |
| | $defs 的key name不能包括 / 字符 | |
diff --git a/keyword_validators.go b/keyword_validators.go
index 5e10952..5957bcd 100644
--- a/keyword_validators.go
+++ b/keyword_validators.go
@@ -47,13 +47,10 @@ func (v *keywordValidators) ValidateProperties(value any, context *validationCon
return context.RaiseErrorWithSimplify("properties must be an object", path, SimplifyRemoveProperties)
}
+ // The empty string is a property name like any other: 2020-12 puts no
+ // constraint on the keys of properties, and the enforcer handles it, down to
+ // generating {"": ...}. It is not checked or rewritten here.
for propName, propSchema := range props {
- if v.config.IsUltra() || v.config.IsTest() {
- if propName == "" {
- return context.RaiseErrorWithSimplify("property name cannot be empty", path, SimplifyDefault)
- }
- }
-
if InvalidPropertyNames[propName] && v.config.IsGreaterThanStrict() {
return context.RaiseErrorWithSimplify(fmt.Sprintf("property name '%s' is reserved for JSON Schema keywords", propName), path, SimplifyRemoveSubSchema)
}
@@ -77,18 +74,13 @@ func (v *keywordValidators) ValidateRequired(value any, context *validationConte
return nil
}
- // Check that all required properties are strings
+ // Requiring the empty-string property is legal and the enforcer supports it, so
+ // only the entry's type is checked here. Whether the property is declared at all
+ // is checked further down, for the empty string like for any other name.
for _, prop := range required {
- propStr, ok := prop.(string)
- if !ok {
+ if _, ok := prop.(string); !ok {
return context.RaiseErrorWithSimplify("items in required array must be strings", path, SimplifyRemoveRequired)
}
-
- if v.config.IsUltra() || v.config.IsTest() {
- if propStr == "" {
- return context.RaiseErrorWithSimplify("property names in required array cannot be empty", path, SimplifyRemoveRequired)
- }
- }
}
// Check for duplicates
@@ -122,31 +114,39 @@ func (v *keywordValidators) ValidateRequired(value any, context *validationConte
// Check if current schema is object type
schemaType, hasType := current[Type]
if !hasType {
- return context.RaiseErrorWithSimplify("required keyword must be used with object", path, SimplifyDefault)
+ return context.RaiseErrorWithSimplify("required keyword must be used with object", path, SimplifyRemoveRequired)
}
switch t := schemaType.(type) {
case string:
if t != Object {
- return context.RaiseErrorWithSimplify("required keyword must be used with object", path, SimplifyDefault)
+ return context.RaiseErrorWithSimplify("required keyword must be used with object", path, SimplifyRemoveRequired)
}
case SchemaList:
if len(t) != 1 || t[0] != Object {
- return context.RaiseErrorWithSimplify("required keyword must be used with object", path, SimplifyDefault)
+ return context.RaiseErrorWithSimplify("required keyword must be used with object", path, SimplifyRemoveRequired)
}
}
// Check if properties exists
if !hasProps {
- return context.RaiseErrorWithSimplify("required specified but 'properties' keyword is missing", path, SimplifyDefault)
+ return context.RaiseErrorWithSimplify("required specified but 'properties' keyword is missing", path, SimplifyRemoveRequired)
}
}
- // Check if all required fields are defined in properties
- for _, prop := range required {
- propStr := prop.(string)
- if _, exists := props[propStr]; !exists {
- return context.RaiseErrorWithSimplify(fmt.Sprintf("required property '%s' is not defined in properties", propStr), path, SimplifyRemoveRequired)
+ // Requiring a property the schema does not describe is legal under 2020-12: the
+ // instance has to carry it, with nothing said about its value. The enforcer has
+ // nothing to generate from that, so only the canonicalising levels report it and
+ // the entry is pruned; looser levels accept the schema as written.
+ if v.config.IsUltra() || v.config.IsTest() {
+ for _, prop := range required {
+ propStr := prop.(string)
+ if _, exists := props[propStr]; !exists {
+ return context.RaiseErrorWithSimplify(
+ fmt.Sprintf("required property '%s' is not defined in properties", propStr),
+ path, SimplifyPruneRequired,
+ )
+ }
}
}
@@ -234,25 +234,31 @@ func (v *keywordValidators) ValidateEnum(value any, context *validationContext,
}
}
- // Validate each enum value matches at least one type in typeList
- for _, val := range enum {
- matchesAnyType := false
- for _, t := range typeList {
- matches, err := v.utils.IsTypeMatch(val, t, context, path)
- if err != nil {
- return err
- }
- if matches {
- matchesAnyType = true
- break
+ // An enum value of the wrong type is legal under 2020-12: type and enum both
+ // apply, so such a value is simply unreachable and the effective constraint is
+ // the intersection. Only the canonicalising levels report it, and the value is
+ // dropped from the enum rather than the enum from the schema; looser levels
+ // accept the schema as written.
+ if v.config.IsUltra() || v.config.IsTest() {
+ for _, val := range enum {
+ matchesAnyType := false
+ for _, t := range typeList {
+ matches, err := v.utils.IsTypeMatch(val, t, context, path)
+ if err != nil {
+ return err
+ }
+ if matches {
+ matchesAnyType = true
+ break
+ }
}
- }
- if !matchesAnyType {
- return context.RaiseErrorWithSimplify(
- fmt.Sprintf("enum value (%v) does not match any type in %v", val, typeList),
- path, SimplifyRemoveEnum,
- )
+ if !matchesAnyType {
+ return context.RaiseErrorWithSimplify(
+ fmt.Sprintf("enum value (%v) does not match any type in %v", val, typeList),
+ path, v.simplifyIntersectEnumWithType(typeList),
+ )
+ }
}
}
@@ -349,24 +355,32 @@ func (v *keywordValidators) ValidateRef(value any, context *validationContext, p
}
// Check if $ref is allowed at the same level as other keywords
- for key := range parentSchema {
- if key == Ref {
- continue
- }
+ if v.config.IsUltra() || v.config.IsTest() {
+ var siblings []string
+ for key := range parentSchema {
+ if key == Ref {
+ continue
+ }
- if is_root && TopLevelOnlyKeywords[key] {
- continue
- }
+ if is_root && TopLevelOnlyKeywords[key] {
+ continue
+ }
- if v.config.IsUltra() || v.config.IsTest() {
if !CommonKeywords[key] {
- return context.RaiseErrorWithSimplify(
- fmt.Sprintf("keyword '%s' is not allowed at the same level as $ref", key),
- path.StringWithoutLast(),
- SimplifyDefault,
- )
+ siblings = append(siblings, key)
}
}
+
+ if len(siblings) > 0 {
+ slices.Sort(siblings)
+ // Drop the offending siblings only. Keeping $ref and the rest of the
+ // schema preserves the referenced constraints instead of losing them all.
+ return context.RaiseErrorWithSimplify(
+ fmt.Sprintf("keyword '%s' is not allowed at the same level as $ref", siblings[0]),
+ path.StringWithoutLast(),
+ SimplifyRemoveSchemaKeys(siblings),
+ )
+ }
}
// Add to ref_paths
@@ -597,7 +611,7 @@ func (v *keywordValidators) ValidateLengthRange(value any, context *validationCo
if minVal > maxVal {
return context.RaiseErrorWithSimplify(
fmt.Sprintf("minLength (%v) cannot be greater than maxLength (%v)", minLength, maxLength),
- path.Append(MinLength), SimplifyRemoveConstraints,
+ path.Append(MinLength), SimplifyDegradeEnclosingSchema,
)
}
}
@@ -685,7 +699,7 @@ func (v *keywordValidators) ValidateNumericRange(value any, context *validationC
if minVal > maxVal {
return context.RaiseErrorWithSimplify(
fmt.Sprintf("minimum (%v) cannot be greater than maximum (%v)", minimum, maximum),
- path, SimplifyRemoveConstraints,
+ path, SimplifyRemoveParentSchema,
)
}
}
@@ -739,7 +753,7 @@ func (v *keywordValidators) ValidateItemsRange(value any, context *validationCon
if minVal > maxVal {
return context.RaiseErrorWithSimplify(
fmt.Sprintf("minItems (%v) cannot be greater than maxItems (%v)", minItems, maxItems),
- path.Append(MinItems), SimplifyRemoveConstraints,
+ path.Append(MinItems), SimplifyDegradeEnclosingSchema,
)
}
}
diff --git a/pyproject.toml b/pyproject.toml
index de06e09..1361dd4 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
[project]
name = "walle"
-version = "0.1.14"
+version = "0.1.15"
description = "Moonshot-flavored JSON schema validator"
readme = "README.md"
requires-python = ">=3.9"
diff --git a/python/walle/lib/libwalle.so b/python/walle/lib/libwalle.so
index 0182148..9a5582d 100644
Binary files a/python/walle/lib/libwalle.so and b/python/walle/lib/libwalle.so differ
diff --git a/refmerge.go b/refmerge.go
new file mode 100644
index 0000000..6aa94f1
--- /dev/null
+++ b/refmerge.go
@@ -0,0 +1,821 @@
+package walle
+
+import (
+ "encoding/json"
+ "fmt"
+ "reflect"
+ "sort"
+)
+
+// Merging sibling constraints into a $ref target.
+
+// lowerBoundKeywords tighten as the value grows.
+var lowerBoundKeywords = map[string]bool{
+ MinLength: true,
+ MinItems: true,
+ Minimum: true,
+}
+
+// upperBoundKeywords tighten as the value shrinks.
+var upperBoundKeywords = map[string]bool{
+ MaxLength: true,
+ MaxItems: true,
+ Maximum: true,
+}
+
+// refInlineBudget tracks the remaining copy allowance for one inlining pass,
+// in serialized bytes. The budget is the headroom left under the schema size
+// limit, so it is always the budget -- not the size check -- that decides when
+// inlining stops; that keeps the degradation local instead of letting the size
+// check discard the whole schema. Once a merge is refused the budget stays
+// closed: the alternative, measuring every later target before refusing it,
+// can itself be driven into quadratic time.
+type refInlineBudget struct {
+ remaining int
+ exhausted bool
+}
+
+func (b *refInlineBudget) tryConsume(cost int) bool {
+ if b.exhausted {
+ return false
+ }
+ if cost > b.remaining {
+ b.exhausted = true
+ return false
+ }
+ b.remaining -= cost
+ return true
+}
+
+// serializedSize approximates the bytes a value adds to the marshaled schema.
+func serializedSize(node any) int {
+ encoded, err := json.Marshal(node)
+ if err != nil {
+ return 0
+ }
+ return len(encoded)
+}
+
+// inlineConflictingRefSiblings rewrites every node that holds both a $ref and
+// its own constraints, replacing it with the merged schema. The input is not
+// modified; when there is nothing to merge the original is returned as is.
+//
+// Inlining copies the definition at every use site, so a DAG of definitions
+// expands exponentially. Once the copy budget -- the headroom under the schema
+// size limit -- runs out, the siblings that could not be folded in are stripped
+// in one pass and reported, so the caller can warn about exactly what was lost.
+func inlineConflictingRefSiblings(root SchemaDict, maxSchemaSize int) (SchemaDict, []string) {
+ if len(root) == 0 || !hasMergeableRefSibling(root, root) {
+ return root, nil
+ }
+
+ copied, ok := deepCopyValue(root).(SchemaDict)
+ if !ok {
+ return root, nil
+ }
+
+ budget := &refInlineBudget{remaining: maxSchemaSize - serializedSize(copied)}
+ mergeRefSiblings(copied, copied, budget)
+ pruneOrphanDefs(copied)
+ if !budget.exhausted {
+ return copied, nil
+ }
+
+ return copied, stripRefSiblings(copied, copied, nil)
+}
+
+// stripRefSiblings removes the sibling constraints of every node that was
+// still mergeable when the copy budget ran out. This is the same drop
+// validation would apply one error at a time; doing it in one pass keeps
+// Canonical from exhausting its retry budget on large schemas and returning
+// {} instead of a loosened schema. Nodes whose siblings contradict the
+// referenced schema are left alone: validation merges those, keeping whatever
+// the two sides agree on, which loses less than a wholesale drop. The result
+// lists every dropped keyword with its location.
+func stripRefSiblings(node any, root SchemaDict, path []string) []string {
+ var dropped []string
+ switch n := node.(type) {
+ case SchemaDict:
+ if ref, ok := n[Ref].(string); ok {
+ if siblings, ok := mergeableSiblings(n, root, ref); ok {
+ keys := make([]string, 0, len(siblings))
+ for key := range siblings {
+ keys = append(keys, key)
+ }
+ sort.Strings(keys)
+
+ where := "root"
+ if len(path) > 0 {
+ where = newSchemaPathFromParts(path).String()
+ }
+ for _, key := range keys {
+ delete(n, key)
+ dropped = append(dropped, fmt.Sprintf("%s at %s", key, where))
+ }
+ }
+ }
+ for key, value := range n {
+ dropped = append(dropped, stripRefSiblings(value, root, append(path, key))...)
+ }
+ case SchemaList:
+ for i, item := range n {
+ dropped = append(dropped, stripRefSiblings(item, root, append(path, fmt.Sprintf("{%d}", i)))...)
+ }
+ }
+ return dropped
+}
+
+// mergeableSiblings returns the constraint keywords a node carries beside its
+// $ref when they could safely have been folded in, mirroring what validation
+// would drop: annotations and the root's own $defs / $id stay, and so does a
+// node whose siblings contradict the referenced schema.
+func mergeableSiblings(node SchemaDict, root SchemaDict, ref string) (SchemaDict, bool) {
+ target, ok := resolveDefsRef(root, ref)
+ if !ok || len(target) == 0 {
+ return nil, false
+ }
+
+ siblings := make(SchemaDict, len(node)-1)
+ for key, value := range node {
+ if key == Ref || CommonKeywords[key] || (TopLevelOnlyKeywords[key] && sameSchemaDict(node, root)) {
+ continue
+ }
+ siblings[key] = value
+ }
+ if len(siblings) == 0 {
+ return nil, false
+ }
+ if unsatisfiableOverlap(siblings, target) != "" {
+ return nil, false
+ }
+ return siblings, true
+}
+
+// distributeAnyOfParentKeywords rewrites every node that constrains an instance
+// both directly and through anyOf. The two apply together, but the enforcer's
+// anyOf is a plain union with no room for a conjunct beside it, so the parent
+// constraint is pushed into each branch instead:
+//
+// {minLength: 5, anyOf: [A, B]} == {anyOf: [A and {minLength: 5}, B and {minLength: 5}]}
+//
+// A branch that cannot agree with the parent is dropped: no instance reaches it.
+// Dropping every branch leaves a node nothing can satisfy, which degrades to {}
+// rather than to an empty anyOf that would accept everything.
+//
+// The input is not modified; when there is nothing to distribute the original is
+// returned as is.
+func distributeAnyOfParentKeywords(root SchemaDict) SchemaDict {
+ if len(root) == 0 || !hasDistributableAnyOf(root, root) {
+ return root
+ }
+
+ copied, ok := deepCopyValue(root).(SchemaDict)
+ if !ok {
+ return root
+ }
+
+ distributeAnyOf(copied, copied)
+ return copied
+}
+
+// hasDistributableAnyOf reports whether the subtree holds a node worth rewriting,
+// so the common case costs a single read-only walk.
+func hasDistributableAnyOf(node any, root SchemaDict) bool {
+ switch n := node.(type) {
+ case SchemaDict:
+ if _, _, ok := distributableAnyOf(n, root); ok {
+ return true
+ }
+ for _, value := range n {
+ if hasDistributableAnyOf(value, root) {
+ return true
+ }
+ }
+ case SchemaList:
+ for _, item := range n {
+ if hasDistributableAnyOf(item, root) {
+ return true
+ }
+ }
+ }
+ return false
+}
+
+// distributeAnyOf walks the tree and rewrites every distributable node in place.
+func distributeAnyOf(node any, root SchemaDict) {
+ switch n := node.(type) {
+ case SchemaDict:
+ if branches, outer, ok := distributableAnyOf(n, root); ok {
+ merged := make(SchemaList, 0, len(branches))
+ for _, branch := range branches {
+ branchDict, ok := branch.(SchemaDict)
+ if !ok {
+ continue
+ }
+ if branchContradictsParent(root, outer, branchDict) {
+ continue
+ }
+ merged = append(merged, mergeStricter(outer, branchDict))
+ }
+
+ for key := range n {
+ if key == AnyOf || !distributableKeyword(key, n, root) {
+ continue
+ }
+ delete(n, key)
+ }
+
+ if len(merged) == 0 {
+ for key := range n {
+ delete(n, key)
+ }
+ } else {
+ n[AnyOf] = merged
+ }
+ }
+ for _, value := range n {
+ distributeAnyOf(value, root)
+ }
+ case SchemaList:
+ for _, item := range n {
+ distributeAnyOf(item, root)
+ }
+ }
+}
+
+// distributableAnyOf returns the branches and the parent constraints to push into
+// them, for a node that carries both. Annotations and the root's own $defs / $id
+// are not constraints and stay where they are.
+func distributableAnyOf(node SchemaDict, root SchemaDict) (SchemaList, SchemaDict, bool) {
+ branches, ok := node[AnyOf].(SchemaList)
+ if !ok || len(branches) == 0 {
+ return nil, nil, false
+ }
+ for _, branch := range branches {
+ if _, ok := branch.(SchemaDict); !ok {
+ return nil, nil, false
+ }
+ }
+
+ outer := make(SchemaDict, len(node))
+ for key, value := range node {
+ if key != AnyOf && distributableKeyword(key, node, root) {
+ outer[key] = value
+ }
+ }
+ if len(outer) == 0 {
+ return nil, nil, false
+ }
+
+ return branches, outer, true
+}
+
+// distributableKeyword reports whether a keyword beside anyOf is a constraint on
+// the instance, as opposed to an annotation or a document-level declaration.
+func distributableKeyword(keyword string, node SchemaDict, root SchemaDict) bool {
+ if CommonKeywords[keyword] {
+ return false
+ }
+ // $defs and $id describe the document, not the instance, and are only legal at
+ // the root, which is exactly where they must be left alone.
+ return !(TopLevelOnlyKeywords[keyword] && sameSchemaDict(node, root))
+}
+
+func sameSchemaDict(a, b SchemaDict) bool {
+ return reflect.ValueOf(a).Pointer() == reflect.ValueOf(b).Pointer()
+}
+
+// branchContradictsParent reports whether no instance can satisfy the branch and
+// the parent constraints at once. A branch that only points at a definition is
+// resolved first, so that a contradiction hidden behind $ref still counts.
+func branchContradictsParent(root SchemaDict, outer SchemaDict, branch SchemaDict) bool {
+ if unsatisfiableOverlap(outer, branch) != "" {
+ return true
+ }
+
+ ref, ok := branch[Ref].(string)
+ if !ok || refIsRecursive(root, ref) {
+ return false
+ }
+ target, ok := resolveDefsRef(root, ref)
+ if !ok {
+ return false
+ }
+ return unsatisfiableOverlap(outer, target) != ""
+}
+
+// pruneOrphanDefs removes root-level $defs entries that nothing points at any
+// more. Inlining leaves such entries behind and they only eat into the size
+// budget, which a large schema cannot afford to waste. Removing one entry can
+// orphan another, so this runs until it settles.
+func pruneOrphanDefs(root SchemaDict) {
+ for pruneOrphanDefsOnce(root) {
+ }
+}
+
+func pruneOrphanDefsOnce(root SchemaDict) bool {
+ defs, ok := root[Defs].(SchemaDict)
+ if !ok || len(defs) == 0 {
+ return false
+ }
+
+ referenced := make(map[string]bool, len(defs))
+ for _, ref := range collectRefs(root) {
+ // "#" resolves to the whole document, so nothing is provably unused.
+ if ref == "#" {
+ return false
+ }
+ parts, ok := defsRefParts(ref)
+ if !ok || len(parts) < 2 || parts[0] != Defs {
+ continue
+ }
+ referenced[parts[1]] = true
+ }
+
+ removed := false
+ for name := range defs {
+ if !referenced[name] {
+ delete(defs, name)
+ removed = true
+ }
+ }
+
+ if len(defs) == 0 {
+ delete(root, Defs)
+ }
+ return removed
+}
+
+// hasMergeableRefSibling reports whether the subtree holds a node worth merging,
+// so the common case costs a single read-only walk.
+func hasMergeableRefSibling(node any, root SchemaDict) bool {
+ switch n := node.(type) {
+ case SchemaDict:
+ if _, _, ok := mergeableRefTarget(n, root); ok {
+ return true
+ }
+ for _, value := range n {
+ if hasMergeableRefSibling(value, root) {
+ return true
+ }
+ }
+ case SchemaList:
+ for _, item := range n {
+ if hasMergeableRefSibling(item, root) {
+ return true
+ }
+ }
+ }
+ return false
+}
+
+// mergeRefSiblings walks the tree and merges every mergeable node in place. A
+// node the budget refuses keeps its $ref and siblings; stripRefSiblings drops
+// them afterwards in one pass, the same fallback validation would apply.
+func mergeRefSiblings(node any, root SchemaDict, budget *refInlineBudget) {
+ switch n := node.(type) {
+ case SchemaDict:
+ if target, siblings, ok := mergeableRefTarget(n, root); ok {
+ // Charge the marginal growth: the merged node replaces the $ref node,
+ // so the node's own bytes come back. Merged output never exceeds
+ // target plus siblings, so this stays an upper bound of the real cost.
+ cost := serializedSize(target) + serializedSize(siblings) - serializedSize(n)
+ if cost < 0 {
+ cost = 0
+ }
+ if budget.tryConsume(cost) {
+ merged := mergeStricter(siblings, target)
+ for key := range n {
+ delete(n, key)
+ }
+ for key, value := range merged {
+ n[key] = value
+ }
+ // The merged result can itself carry a $ref from the definition body,
+ // but its siblings have already been folded in, so keep walking the
+ // children only.
+ }
+ }
+ for _, value := range n {
+ mergeRefSiblings(value, root, budget)
+ }
+ case SchemaList:
+ for _, item := range n {
+ mergeRefSiblings(item, root, budget)
+ }
+ }
+}
+
+// mergeableRefTarget returns the referenced schema and the sibling constraints
+// when node is a $ref carrying its own constraints that can safely be folded in.
+func mergeableRefTarget(node SchemaDict, root SchemaDict) (SchemaDict, SchemaDict, bool) {
+ ref, ok := node[Ref].(string)
+ if !ok || len(node) < 2 {
+ return nil, nil, false
+ }
+
+ siblings := make(SchemaDict, len(node)-1)
+ for key, value := range node {
+ if key != Ref {
+ siblings[key] = value
+ }
+ }
+ if len(siblings) == 0 {
+ return nil, nil, false
+ }
+
+ target, ok := resolveDefsRef(root, ref)
+ if !ok || len(target) == 0 {
+ return nil, nil, false
+ }
+
+ // Only step in when a real constraint would otherwise be lost: the definition
+ // either does not carry the keyword at all, or carries a different value.
+ // Annotations and exact duplicates are left to the ordinary simplify path,
+ // which strips the outer copy without changing what the schema accepts.
+ if !siblingsWouldLoseConstraint(siblings, target) {
+ return nil, nil, false
+ }
+
+ // A contradiction is not something merging can fix; leave the node intact so
+ // that validation reports it as unsatisfiable.
+ if unsatisfiableOverlap(siblings, target) != "" {
+ return nil, nil, false
+ }
+
+ if refIsRecursive(root, ref) {
+ return nil, nil, false
+ }
+
+ return target, siblings, true
+}
+
+// mergeRefSiblingDroppingContradiction folds the siblings of node into its $ref
+// target for the nodes mergeableRefTarget refuses to touch because the two sides
+// contradict each other. Whatever they still agree on survives and only the
+// contradicting keyword goes, which leaves the node unconstrained in that one
+// dimension instead of asserting one of the two sides.
+//
+// Two sides that cannot agree on the type are the exception. Every other keyword
+// only means anything relative to a type, so combining a string bound from one
+// side with a numeric bound from the other produces a schema that describes
+// nothing; there the caller is told to keep none of it.
+func mergeRefSiblingDroppingContradiction(root, node SchemaDict) (SchemaDict, bool) {
+ ref, ok := node[Ref].(string)
+ if !ok {
+ return nil, false
+ }
+
+ target, ok := resolveDefsRef(root, ref)
+ if !ok || len(target) == 0 || refIsRecursive(root, ref) {
+ return nil, false
+ }
+
+ siblings := make(SchemaDict, len(node)-1)
+ for key, value := range node {
+ if key != Ref {
+ siblings[key] = value
+ }
+ }
+ if len(siblings) == 0 {
+ return nil, false
+ }
+
+ if typeSetsDisjoint(typeSet(siblings[Type]), typeSet(target[Type])) {
+ return nil, false
+ }
+
+ return mergeStricter(siblings, target), true
+}
+
+// siblingsWouldLoseConstraint reports whether dropping the siblings would weaken
+// the schema.
+func siblingsWouldLoseConstraint(siblings, target SchemaDict) bool {
+ for key, value := range siblings {
+ if CommonKeywords[key] {
+ continue
+ }
+ existing, ok := target[key]
+ if !ok || !reflect.DeepEqual(existing, value) {
+ return true
+ }
+ }
+ return false
+}
+
+// resolveDefsRef resolves "#/$defs/Name" style pointers against root. "#" is
+// deliberately unsupported: it always resolves to the whole document and is
+// therefore recursive by construction.
+func resolveDefsRef(root SchemaDict, ref string) (SchemaDict, bool) {
+ parts, ok := defsRefParts(ref)
+ if !ok {
+ return nil, false
+ }
+
+ current := root
+ for _, part := range parts {
+ next, ok := current[part].(SchemaDict)
+ if !ok {
+ return nil, false
+ }
+ current = next
+ }
+ return current, true
+}
+
+// defsRefParts splits "#/$defs/A/$defs/B" into its dictionary keys.
+func defsRefParts(ref string) ([]string, bool) {
+ const prefix = "#/"
+ if len(ref) <= len(prefix) || ref[:len(prefix)] != prefix {
+ return nil, false
+ }
+
+ var parts []string
+ start := len(prefix)
+ for i := start; i <= len(ref); i++ {
+ if i == len(ref) || ref[i] == '/' {
+ if i == start {
+ return nil, false
+ }
+ parts = append(parts, ref[start:i])
+ start = i + 1
+ }
+ }
+ if len(parts) == 0 {
+ return nil, false
+ }
+ return parts, true
+}
+
+// refIsRecursive reports whether following ref can lead back to ref itself,
+// directly or through other definitions.
+func refIsRecursive(root SchemaDict, ref string) bool {
+ seen := map[string]bool{ref: true}
+ queue := []string{ref}
+
+ for len(queue) > 0 {
+ current := queue[0]
+ queue = queue[1:]
+
+ target, ok := resolveDefsRef(root, current)
+ if !ok {
+ continue
+ }
+
+ for _, next := range collectRefs(target) {
+ if next == ref {
+ return true
+ }
+ // "#" pulls in the whole document, definitions included, so treat any
+ // definition that reaches it as recursive.
+ if next == "#" {
+ return true
+ }
+ if !seen[next] {
+ seen[next] = true
+ queue = append(queue, next)
+ }
+ }
+ }
+ return false
+}
+
+// collectRefs gathers every $ref string in the subtree.
+func collectRefs(node any) []string {
+ var refs []string
+ switch n := node.(type) {
+ case SchemaDict:
+ if ref, ok := n[Ref].(string); ok {
+ refs = append(refs, ref)
+ }
+ for _, value := range n {
+ refs = append(refs, collectRefs(value)...)
+ }
+ case SchemaList:
+ for _, item := range n {
+ refs = append(refs, collectRefs(item)...)
+ }
+ }
+ return refs
+}
+
+// mergeStricter folds the sibling constraints into a copy of the referenced
+// definition, keeping the stricter value wherever both sides set a keyword.
+func mergeStricter(siblings, target SchemaDict) SchemaDict {
+ merged := make(SchemaDict, len(siblings)+len(target))
+ for key, value := range target {
+ merged[key] = deepCopyValue(value)
+ }
+
+ for key, value := range siblings {
+ existing, clash := merged[key]
+ if !clash {
+ merged[key] = deepCopyValue(value)
+ continue
+ }
+ if stricter, keep := stricterValue(key, existing, deepCopyValue(value)); keep {
+ merged[key] = stricter
+ } else {
+ delete(merged, key)
+ }
+ }
+
+ return merged
+}
+
+// stricterValue picks whichever of the two values admits fewer instances.
+// fromTarget comes from the definition, fromSibling from the use site. A false
+// second return means the two sides have nothing in common: there is no value to
+// keep, so the caller drops the keyword rather than picking a side and accepting
+// what the other side ruled out.
+func stricterValue(keyword string, fromTarget, fromSibling any) (any, bool) {
+ switch {
+ case lowerBoundKeywords[keyword]:
+ return largerNumber(fromTarget, fromSibling), true
+ case upperBoundKeywords[keyword]:
+ return smallerNumber(fromTarget, fromSibling), true
+ case keyword == Type:
+ return intersectTypes(fromTarget, fromSibling)
+ case keyword == Enum:
+ return intersectEnums(fromTarget, fromSibling)
+ case keyword == Required:
+ return unionRequired(fromTarget, fromSibling), true
+ case keyword == Properties:
+ return mergeProperties(fromTarget, fromSibling), true
+ }
+
+ // Anything else, annotations included, keeps the use site's value: it is the
+ // more specific of the two. Two different patterns cannot be intersected into
+ // one pattern, so the same rule applies there.
+ return fromSibling, true
+}
+
+func largerNumber(a, b any) any {
+ x, okA := numericValue(a)
+ y, okB := numericValue(b)
+ if !okA {
+ return b
+ }
+ if !okB {
+ return a
+ }
+ if x >= y {
+ return a
+ }
+ return b
+}
+
+func smallerNumber(a, b any) any {
+ x, okA := numericValue(a)
+ y, okB := numericValue(b)
+ if !okA {
+ return b
+ }
+ if !okB {
+ return a
+ }
+ if x <= y {
+ return a
+ }
+ return b
+}
+
+func numericValue(value any) (float64, bool) {
+ switch v := value.(type) {
+ case float64:
+ return v, true
+ case float32:
+ return float64(v), true
+ case int:
+ return float64(v), true
+ case int64:
+ return float64(v), true
+ }
+ return 0, false
+}
+
+// intersectTypes keeps the types allowed by both sides. integer wins over number
+// because every integer is a number but not the other way round.
+func intersectTypes(fromTarget, fromSibling any) (any, bool) {
+ setTarget, setSibling := typeSet(fromTarget), typeSet(fromSibling)
+ if len(setTarget) == 0 {
+ return fromSibling, true
+ }
+ if len(setSibling) == 0 {
+ return fromTarget, true
+ }
+
+ var names []string
+ for name := range setTarget {
+ if _, ok := setSibling[name]; ok {
+ names = append(names, name)
+ continue
+ }
+ // integer is the narrower side of an integer/number overlap
+ switch name {
+ case Number:
+ if _, ok := setSibling[Integer]; ok {
+ names = append(names, Integer)
+ }
+ case Integer:
+ if _, ok := setSibling[Number]; ok {
+ names = append(names, Integer)
+ }
+ }
+ }
+
+ if len(names) == 0 {
+ return nil, false
+ }
+
+ sort.Strings(names)
+ if len(names) == 1 {
+ return names[0], true
+ }
+ list := make(SchemaList, len(names))
+ for i, name := range names {
+ list[i] = name
+ }
+ return list, true
+}
+
+// intersectEnums keeps the values allowed by both sides.
+func intersectEnums(fromTarget, fromSibling any) (any, bool) {
+ listTarget, okTarget := fromTarget.(SchemaList)
+ listSibling, okSibling := fromSibling.(SchemaList)
+ if !okTarget || !okSibling {
+ return fromSibling, true
+ }
+
+ var out SchemaList
+ for _, candidate := range listTarget {
+ for _, allowed := range listSibling {
+ if reflect.DeepEqual(candidate, allowed) {
+ out = append(out, candidate)
+ break
+ }
+ }
+ }
+
+ if len(out) == 0 {
+ return nil, false
+ }
+ return out, true
+}
+
+// unionRequired keeps every property either side insists on.
+func unionRequired(fromTarget, fromSibling any) any {
+ listTarget, _ := fromTarget.(SchemaList)
+ listSibling, _ := fromSibling.(SchemaList)
+
+ seen := make(map[string]bool, len(listTarget)+len(listSibling))
+ out := make(SchemaList, 0, len(listTarget)+len(listSibling))
+ for _, list := range []SchemaList{listTarget, listSibling} {
+ for _, item := range list {
+ name, ok := item.(string)
+ if !ok || seen[name] {
+ continue
+ }
+ seen[name] = true
+ out = append(out, name)
+ }
+ }
+
+ if len(out) == 0 {
+ return fromSibling
+ }
+ return out
+}
+
+// mergeProperties keeps properties from both sides. A property both sides
+// describe is merged the same way as any other subschema, so neither side's
+// constraints are dropped; letting the use site simply win would lose whatever
+// the definition asserted about that property.
+func mergeProperties(fromTarget, fromSibling any) any {
+ dictTarget, okTarget := fromTarget.(SchemaDict)
+ dictSibling, okSibling := fromSibling.(SchemaDict)
+ if !okTarget || !okSibling {
+ return fromSibling
+ }
+
+ out := make(SchemaDict, len(dictTarget)+len(dictSibling))
+ for name, value := range dictTarget {
+ out[name] = value
+ }
+ for name, value := range dictSibling {
+ existing, clash := out[name]
+ if !clash {
+ out[name] = value
+ continue
+ }
+
+ targetProperty, okT := existing.(SchemaDict)
+ siblingProperty, okS := value.(SchemaDict)
+ if !okT || !okS {
+ out[name] = value
+ continue
+ }
+ out[name] = mergeStricter(siblingProperty, targetProperty)
+ }
+ return out
+}
diff --git a/simplifiers.go b/simplifiers.go
index d485ac0..a12865a 100644
--- a/simplifiers.go
+++ b/simplifiers.go
@@ -14,53 +14,70 @@ func extractSubSchema(schema Schema, path schemaPath) (Schema, error) {
parts := path.Parts
invalidPathErr := fmt.Errorf("invalid path: %s", path.String())
- if len(parts) > 1 {
- for _, part := range parts {
- // Handle anyOf{index} pattern
- if strings.Contains(part, "{") && strings.Contains(part, "}") {
- baseParts := strings.Split(part, "{")
- if len(baseParts) < 2 {
- return nil, invalidPathErr
- }
- base := baseParts[0]
- schemaIndex, err := strconv.Atoi(strings.TrimSuffix(baseParts[1], "}"))
- if err != nil {
- return nil, invalidPathErr
- }
- baseValue, exists := current[base]
- if !exists {
- return nil, invalidPathErr
- }
-
- currentList, ok := baseValue.(SchemaList)
- if !ok {
- return nil, invalidPathErr
- }
- if schemaIndex < 0 || schemaIndex >= len(currentList) {
- return nil, invalidPathErr
- }
-
- itemDict, ok := currentList[schemaIndex].(SchemaDict)
- if !ok {
- return nil, invalidPathErr
- }
-
- current = itemDict
+ // A single plain part names the keyword the error is about, so the schema
+ // holding it is the root. An indexed part such as "anyOf{1}" names a branch
+ // instead, and has to be descended into even when it stands alone.
+ if len(parts) == 1 && !isIndexedPathPart(parts[0]) {
+ return current, nil
+ }
- } else {
- if next, ok := current[part].(SchemaDict); ok {
- current = next
- } else {
- return nil, fmt.Errorf("invalid path: %s", path.String())
- }
+ for _, part := range parts {
+ if isIndexedPathPart(part) {
+ next, err := resolveIndexedPathPart(current, part)
+ if err != nil {
+ return nil, invalidPathErr
}
+ current = next
+ continue
+ }
+
+ next, ok := current[part].(SchemaDict)
+ if !ok {
+ return nil, invalidPathErr
}
+ current = next
}
return current, nil
}
+// isIndexedPathPart reports whether a path part addresses a list entry, as
+// "anyOf{1}" does.
+func isIndexedPathPart(part string) bool {
+ return strings.Contains(part, "{") && strings.Contains(part, "}")
+}
+
+// resolveIndexedPathPart follows a part such as "anyOf{1}" into the list entry it
+// names.
+func resolveIndexedPathPart(current Schema, part string) (Schema, error) {
+ invalidPathErr := fmt.Errorf("invalid path part: %s", part)
+
+ baseParts := strings.Split(part, "{")
+ if len(baseParts) < 2 {
+ return nil, invalidPathErr
+ }
+
+ index, err := strconv.Atoi(strings.TrimSuffix(baseParts[1], "}"))
+ if err != nil {
+ return nil, invalidPathErr
+ }
+
+ list, ok := current[baseParts[0]].(SchemaList)
+ if !ok {
+ return nil, invalidPathErr
+ }
+ if index < 0 || index >= len(list) {
+ return nil, invalidPathErr
+ }
+
+ entry, ok := list[index].(SchemaDict)
+ if !ok {
+ return nil, invalidPathErr
+ }
+ return entry, nil
+}
+
func removeAtPath(schema Schema, path schemaPath, targetKey string, mustExist bool) error {
current, err := extractSubSchema(schema, path)
if err != nil {
@@ -120,6 +137,24 @@ func SimplifyDefault(schema Schema, _ schemaPath) Schema {
return schema
}
+// resolveDictAtPath returns the dict that path points at. extractSubSchema treats a
+// single-part path as the root, which is wrong for paths like "properties", so that
+// case is resolved here explicitly. An indexed part such as "anyOf{0}" already names
+// a dict of its own and is left to extractSubSchema even when it stands alone.
+func resolveDictAtPath(schema Schema, path schemaPath) (Schema, error) {
+ if len(path.Parts) > 1 || (!path.IsRoot() && isIndexedPathPart(path.Last())) {
+ return extractSubSchema(schema, path)
+ }
+ if path.IsRoot() {
+ return schema, nil
+ }
+ next, ok := schema[path.Last()].(SchemaDict)
+ if !ok {
+ return nil, fmt.Errorf("invalid path: %s", path.String())
+ }
+ return next, nil
+}
+
// deletes the given keys from the schema at path
func SimplifyRemoveSchemaKeys(keys []string) SimplifyFunc {
keysCopy := append([]string(nil), keys...)
@@ -135,6 +170,19 @@ func SimplifyRemoveSchemaKeys(keys []string) SimplifyFunc {
}
}
+// simplifyFuncForAnyOfParentConflicts picks how to resolve a keyword that a node
+// states both directly and inside its anyOf branches. A real constraint is pushed
+// into the branches so that the stricter of the two values survives; an annotation
+// carries no constraint to preserve, so the outer copy is simply dropped.
+func simplifyFuncForAnyOfParentConflicts(conflicts []string) SimplifyFunc {
+ for _, keyword := range conflicts {
+ if !CommonKeywords[keyword] {
+ return SimplifyDistributeAnyOfParent
+ }
+ }
+ return SimplifyRemoveSchemaKeys(conflicts)
+}
+
// Pure description/title conflicts remove only those keys at the error path (outer layer).
func simplifyFuncForKeywordConflicts(conflicts []string) SimplifyFunc {
commonKeys := make([]string, 0, len(conflicts))
@@ -175,6 +223,82 @@ func SimplifyRemoveRequired(schema Schema, path schemaPath) Schema {
return schema
}
+// SimplifyPruneRequired drops the entries of required that name no declared
+// property. Demanding a property the schema says nothing about is legal but
+// leaves the enforcer nothing to generate, and removing the whole keyword would
+// also release the properties that are declared and genuinely required.
+func SimplifyPruneRequired(schema Schema, path schemaPath) Schema {
+ holder, err := resolveDictAtPath(schema, path.Parent())
+ if err != nil {
+ return make(Schema)
+ }
+
+ required, ok := holder[Required].(SchemaList)
+ if !ok {
+ delete(holder, Required)
+ return schema
+ }
+
+ props, _ := holder[Properties].(SchemaDict)
+ kept := make(SchemaList, 0, len(required))
+ for _, entry := range required {
+ name, ok := entry.(string)
+ if !ok {
+ continue
+ }
+ if _, declared := props[name]; declared {
+ kept = append(kept, name)
+ }
+ }
+
+ if len(kept) == 0 {
+ delete(holder, Required)
+ return schema
+ }
+ holder[Required] = kept
+ return schema
+}
+
+// simplifyIntersectEnumWithType keeps only the enum values the declared type
+// admits. Those are the ones an instance could ever hold; the rest are already
+// unreachable, so dropping them changes nothing about what the schema accepts.
+//
+// No value surviving means the two keywords leave nothing to satisfy them, and one
+// of them has to give. The enum goes and the type stays, because that is the
+// tightest schema still expressible here: keeping the enum instead would accept
+// values of a type the schema ruled out.
+func (v *keywordValidators) simplifyIntersectEnumWithType(typeList []string) SimplifyFunc {
+ return func(schema Schema, path schemaPath) Schema {
+ holder, err := resolveDictAtPath(schema, path.Parent())
+ if err != nil {
+ return make(Schema)
+ }
+
+ enum, ok := holder[Enum].(SchemaList)
+ if !ok {
+ delete(holder, Enum)
+ return schema
+ }
+
+ kept := make(SchemaList, 0, len(enum))
+ for _, val := range enum {
+ for _, t := range typeList {
+ if matches, err := v.utils.IsTypeMatch(val, t, nil, path); err == nil && matches {
+ kept = append(kept, val)
+ break
+ }
+ }
+ }
+
+ if len(kept) == 0 {
+ delete(holder, Enum)
+ return schema
+ }
+ holder[Enum] = kept
+ return schema
+ }
+}
+
func SimplifyRemoveEnum(schema Schema, path schemaPath) Schema {
err := removeAtPath(schema, path.Parent(), Enum, true)
if err != nil {
@@ -227,6 +351,14 @@ func SimplifyRemovePattern(schema Schema, path schemaPath) Schema {
return schema
}
+// SimplifyDegradeEnclosingSchema empties the schema that holds the keyword at
+// path. Used when a keyword combination is unsatisfiable: dropping the individual
+// bounds would turn "impossible" into "anything goes", which is a far bigger
+// change in meaning than leaving the field unconstrained on purpose.
+func SimplifyDegradeEnclosingSchema(schema Schema, path schemaPath) Schema {
+ return SimplifyRemoveParentSchema(schema, path.Parent())
+}
+
func SimplifyRemoveConstraints(schema Schema, path schemaPath) Schema {
constraints := []string{MinLength, MaxLength, Minimum, Maximum, MinItems, MaxItems}
@@ -298,6 +430,59 @@ func SimplifyRemoveParentSchema(schema Schema, path schemaPath) Schema {
return schema
}
+// SimplifyDistributeAnyOfParent pushes the constraints sitting beside anyOf into
+// each of its branches, where the enforcer can act on them, and drops the branches
+// that cannot agree with them. Deleting the parent constraints instead would let
+// the node accept what they ruled out.
+//
+// A node that cannot be rewritten this way -- a malformed anyOf, for instance --
+// is emptied. Returning it untouched would leave the retry loop reporting the same
+// error until it gives up and throws away the whole document.
+func SimplifyDistributeAnyOfParent(schema Schema, path schemaPath) Schema {
+ node, err := resolveDictAtPath(schema, path)
+ if err != nil {
+ return make(Schema)
+ }
+
+ if _, _, ok := distributableAnyOf(SchemaDict(node), SchemaDict(schema)); !ok {
+ for key := range node {
+ delete(node, key)
+ }
+ return schema
+ }
+
+ distributeAnyOf(SchemaDict(node), SchemaDict(schema))
+ return schema
+}
+
+// SimplifyDropContradictingRefSibling replaces the node at path with the merge of
+// its own constraints and the schema it references, minus the keyword the two
+// sides disagree on. Keeping the rest is the point: a node whose enum cannot
+// overlap the definition's usually still agrees on the type, and emptying the
+// whole node would throw that agreement away too. When the reference cannot be
+// inlined at all, a recursive definition for instance, there is nothing to keep
+// and the node is emptied.
+func SimplifyDropContradictingRefSibling(schema Schema, path schemaPath) Schema {
+ node, err := extractSubSchema(schema, path)
+ if err != nil {
+ return make(Schema)
+ }
+
+ merged, ok := mergeRefSiblingDroppingContradiction(SchemaDict(schema), SchemaDict(node))
+ if !ok {
+ return SimplifyRemoveParentSchema(schema, path)
+ }
+
+ for key := range node {
+ delete(node, key)
+ }
+ for key, value := range merged {
+ node[key] = value
+ }
+
+ return schema
+}
+
func SimplifyRemoveSubSchema(schema Schema, path schemaPath) Schema {
current := schema
parts := path.Parts
diff --git a/testdata/validator_cases/TestBasicTypes/invalid.jsonl b/testdata/validator_cases/TestBasicTypes/invalid.jsonl
index 0a22713..94e79a4 100644
--- a/testdata/validator_cases/TestBasicTypes/invalid.jsonl
+++ b/testdata/validator_cases/TestBasicTypes/invalid.jsonl
@@ -11,7 +11,6 @@
{"schema":"{\"type\": [123, \"string\"]}","expect":"invalid type","unmarshal":false}
{"schema":"\"\"","expect":"JSON schema parsing error","unmarshal":true}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": \"not an object\"\n\t\t\t}","expect":"properties must be an object","unmarshal":false}
-{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"\": {\"type\": \"string\"}\n\t\t\t\t}\n\t\t\t}","expect":"property name cannot be empty","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"user\": {\n\t\t\t\t\t\t\"type\": \"object\",\n\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\ttrue: {\"type\": \"string\"}\n\t\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"JSON schema parsing error","unmarshal":true}
{"schema":"{\n\t\t\t\t\"anyOf\": [\n\t\t\t\t\t{\n\t\t\t\t\t\t\"type\": \"object\",\n\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\t3.14: {\"type\": \"string\"}\n\t\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t]\n\t\t\t}","expect":"JSON schema parsing error","unmarshal":true}
{"schema":"{\n\t\t\t\t\"$defs\": {\n\t\t\t\t\t\"User\": {\n\t\t\t\t\t\t\"type\": \"object\",\n\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\tnull: {\"type\": \"string\"}\n\t\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"JSON schema parsing error","unmarshal":true}
diff --git a/testdata/validator_cases/TestBasicTypes/valid.jsonl b/testdata/validator_cases/TestBasicTypes/valid.jsonl
index 23ee7ed..5ee0418 100644
--- a/testdata/validator_cases/TestBasicTypes/valid.jsonl
+++ b/testdata/validator_cases/TestBasicTypes/valid.jsonl
@@ -24,3 +24,4 @@
{"enum":[0,1,-1,3.14,2.71828],"type":"number"}
{"enum":[true,false],"type":"boolean"}
{"enum":["value",null],"type":["string","null"]}
+{"properties":{"":{"type":"string"}},"type":"object"}
diff --git a/testdata/validator_cases/TestReferences/invalid.jsonl b/testdata/validator_cases/TestReferences/invalid.jsonl
index 4e241b6..d468051 100644
--- a/testdata/validator_cases/TestReferences/invalid.jsonl
+++ b/testdata/validator_cases/TestReferences/invalid.jsonl
@@ -3,7 +3,7 @@
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"user\": {\n\t\t\t\t\t\t\"$ref\": {}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"$ref must be a string","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"user\": {\n\t\t\t\t\t\t\"$ref\": true\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"$ref must be a string","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"user\": {\n\t\t\t\t\t\t\"$ref\": null\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"$ref must be a string","unmarshal":false}
-{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"user\": {\n\t\t\t\t\t\t\"$ref\": \"#\",\n\t\t\t\t\t\t\"type\": \"string\"\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"when using $ref, type should be defined in the referenced schema instead of the parent schema","unmarshal":false}
+{"schema": "{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"user\": {\n\t\t\t\t\t\t\"$ref\": \"#\",\n\t\t\t\t\t\t\"type\": \"string\"\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}", "expect": "the intersection is empty, so no instance can satisfy both", "unmarshal": false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"user\": {\n\t\t\t\t\t\t\"$ref\": \"#/invalid/path\"\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"references must start with #/$defs/","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"user\": {\n\t\t\t\t\t\t\"$ref\": \"#/$defs/User\",\n\t\t\t\t\t\t\"minLength\": 10\n\t\t\t\t\t}\n\t\t\t\t},\n\t\t\t\t\"$defs\": {\n\t\t\t\t\t\"User\": {\n\t\t\t\t\t\t\"type\": \"object\",\n\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\t\"name\": {\"type\": \"string\"}\n\t\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"not allowed at the same level as $ref","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"user\": {\n\t\t\t\t\t\t\"$ref\": \"#/$defs/User\",\n\t\t\t\t\t\t\"minItems\": 10\n\t\t\t\t\t}\n\t\t\t\t},\n\t\t\t\t\"$defs\": {\n\t\t\t\t\t\"User\": {\n\t\t\t\t\t\t\"type\": \"string\"\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"not allowed at the same level as $ref","unmarshal":false}
@@ -15,14 +15,14 @@
{"schema":"{\n\t\t \"type\": \"object\",\n\t\t \"properties\": {\n\t\t \"root\": { \"$ref\": \"#/$defs/root\" }\n\t\t },\n\t\t \"required\": [\"root\"],\n\t\t \"$defs\": {\n\t\t \"root\": {\n\t\t \"type\": \"object\",\n\t\t \"properties\": {\n\t\t \"arrayProp\": {\n\t\t \"type\": \"array\",\n\t\t \"items\": { \"$ref\": \"#/$defs/arrayItem\" }\n\t\t },\n\t\t \"objectProp\": { \"$ref\": \"#/$defs/objectProp\" }\n\t\t },\n\t\t \"required\": [\"arrayProp\", \"objectProp\"]\n\t\t },\n\t\t \"arrayItem\": {\n\t\t \"type\": \"object\",\n\t\t \"properties\": {\n\t\t \"nestedArray\": {\n\t\t \"type\": \"array\",\n\t\t \"items\": { \"$ref\": \"#/$defs/nestedArrayItem\" }\n\t\t }\n\t\t },\n\t\t \"required\": [\"nestedArray\"]\n\t\t },\n\t\t \"nestedArrayItem\": {\n\t\t \"type\": \"object\",\n\t\t \"properties\": {\n\t\t \"root\": { \"$ref\": \"#/$defs/root\" }\n\t\t },\n\t\t \"required\": [\"root\"]\n\t\t },\n\t\t \"objectProp\": {\n\t\t \"type\": \"object\",\n\t\t \"properties\": {\n\t\t \"nestedObject\": { \"$ref\": \"#/$defs/nestedObject\" }\n\t\t },\n\t\t \"required\": [\"nestedObject\"]\n\t\t },\n\t\t \"nestedObject\": {\n\t\t \"type\": \"object\",\n\t\t \"properties\": {\n\t\t \"root\": { \"$ref\": \"#/$defs/root\" }\n\t\t },\n\t\t \"required\": [\"root\"]\n\t\t }\n\t\t }\n\t\t }","expect":"infinite recursion","unmarshal":false}
{"schema":"{\n\t\t \"type\": \"object\",\n\t\t \"properties\": {\n\t\t \"value\": {\n\t\t \"$ref\": \"#/$defs/Node/anyOf\"\n\t\t }\n\t\t },\n\t\t \"$defs\": {\n\t\t \"Node\": {\n\t\t \"anyOf\": [{\"type\": \"string\"}, {\"type\": \"number\"}]\n\t\t }\n\t\t }\n\t\t }","expect":"invalid $ref path","unmarshal":false}
{"schema":"{\n\t\t \"type\": \"object\",\n\t\t \"properties\": {\n\t\t \"value\": {\n\t\t \"$ref\": \"#/$defs/Node\",\n\t\t \"anyOf\": [{\"type\": \"string\"}, {\"type\": \"number\"}]\n\t\t }\n\t\t },\n\t\t \"$defs\": {\n\t\t \"Node\": {\n\t\t \"anyOf\": [{\"type\": \"string\"}, {\"type\": \"number\"}]\n\t\t }\n\t\t }\n\t\t }","expect":"not allowed at the same level","unmarshal":false}
-{"schema":"{\n\t\t\t\t\"type\": [\"object\"],\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"root\": {\n\t\t\t\t\t\t\"$ref\": \"#/$defs/Level1\"\n\t\t\t\t\t}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"root\"],\n\t\t\t\t\"$defs\": {\n\t\t\t\t\t\"Level1\": {\n\t\t\t\t\t\t\"type\": [\"object\"],\n\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\t\"level2\": {\n\t\t\t\t\t\t\t\t\"$ref\": \"#/$defs/Level2\"\n\t\t\t\t\t\t\t}\n\t\t\t\t\t\t},\n\t\t\t\t\t\t\"required\": [\"level2\"]\n\t\t\t\t\t},\n\t\t\t\t\t\"Level2\": {\n\t\t\t\t\t\t\"type\": [\"array\"],\n\t\t\t\t\t\t\"items\": {\n\t\t\t\t\t\t\t\"type\": [\"object\"],\n\t\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\t\t\"level3\": {\n\t\t\t\t\t\t\t\t\t\"$ref\": \"#/$defs/Level3\"\n\t\t\t\t\t\t\t\t}\n\t\t\t\t\t\t\t},\n\t\t\t\t\t\t\t\"required\": [\"level3\"]\n\t\t\t\t\t\t}\n\t\t\t\t\t},\n\t\t\t\t\t\"Level3\": {\n\t\t\t\t\t\t\"type\": [\"object\"],\n\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\t\"level1\": {\n\t\t\t\t\t\t\t\t\"$ref\": \"#/$defs/Level1\"\n\t\t\t\t\t\t\t}\n\t\t\t\t\t\t},\n\t\t\t\t\t\t\"required\": [\"level1\"]\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"infinite recursion","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": [\"object\"],\n\t\t\t\t\"properties\": {\n\t\t\t\t \"node\": {\n\t\t\t\t\t\"$ref\": \"#/$defs/InfiniteNode\"\n\t\t\t\t }\n\t\t\t\t},\n\t\t\t\t\"required\": [\"node\"],\n\t\t\t\t\"$defs\": {\n\t\t\t\t \"InfiniteNode\": {\n\t\t\t\t\t\"type\": [\"object\"],\n\t\t\t\t\t\"properties\": {\n\t\t\t\t\t \"next\": {\n\t\t\t\t\t\t\"$ref\": \"#/$defs/InfiniteNode\"\n\t\t\t\t\t }\n\t\t\t\t\t},\n\t\t\t\t\t\"required\": [\"next\"]\n\t\t\t\t }\n\t\t\t\t}\n\t\t\t}","expect":"infinite recursion","unmarshal":false}
-{"schema":"{\n\t\t\t\t\"type\": [\"object\"],\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"data\": {\n\t\t\t\t\t\"$ref\": \"#/$defs/InfiniteArray\"\n\t\t\t\t\t}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"data\"],\n\t\t\t\t\"$defs\": {\n\t\t\t\t\t\"InfiniteArray\": {\n\t\t\t\t\t\"type\": [\"array\"],\n\t\t\t\t\t\"items\": {\n\t\t\t\t\t\t\"type\": [\"object\"],\n\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\t\"element\": {\n\t\t\t\t\t\t\t\t\"$ref\": \"#/$defs/InfiniteArray\"\n\t\t\t\t\t\t\t}\n\t\t\t\t\t\t},\n\t\t\t\t\t\t\"required\": [\"element\"]\n\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}","expect":"infinite recursion","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"node\": {\n\t\t\t\t\t\t\"anyOf\": [\n\t\t\t\t\t\t\t{\"$ref\": \"#\"},\n\t\t\t\t\t\t\t{\n\t\t\t\t\t\t\t\t\"type\": \"object\",\n\t\t\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\t\t\t\"next\": {\"$ref\": \"#\"}\n\t\t\t\t\t\t\t\t},\n\t\t\t\t\t\t\t\t\"required\": [\"next\"]\n\t\t\t\t\t\t\t}\n\t\t\t\t\t\t]\n\t\t\t\t\t}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"node\"]\n\t\t\t}","expect":"infinite recursion","unmarshal":false}
-{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"name\": {\"type\": \"string\"},\n\t\t\t\t\t\"children\": {\n\t\t\t\t\t\t\"type\": \"array\",\n\t\t\t\t\t\t\"items\": {\"$ref\": \"#\"}\n\t\t\t\t\t}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"children\"]\n\t\t\t}","expect":"infinite recursion","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"node\": {\"$ref\": \"#\"}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"node\"]\n\t\t\t}","expect":"infinite recursion","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"data\": {\n\t\t\t\t\t\"type\": \"object\",\n\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\"value\": {\"type\": \"integer\"},\n\t\t\t\t\t\t\"next\": {\"$ref\": \"#\"}\n\t\t\t\t\t},\n\t\t\t\t\t\"required\": [\"next\"]\n\t\t\t\t\t}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"data\"]\n\t\t\t}","expect":"infinite recursion","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"array\",\n\t\t\t\t\"items\": {\n\t\t\t\t\t\"type\": \"object\",\n\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\"elements\": {\"$ref\": \"#\"}\n\t\t\t\t\t},\n\t\t\t\t\t\"required\": [\"elements\"]\n\t\t\t\t},\n\t\t\t\t\"minItems\": 1\n\t\t\t}","expect":"infinite recursion","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"level1\": {\n\t\t\t\t\t\t\"type\": \"object\",\n\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\t\"level2\": {\n\t\t\t\t\t\t\t\t\"type\": \"object\",\n\t\t\t\t\t\t\t\t\"properties\": {\n\t\t\t\t\t\t\t\t\t\"back\": {\"$ref\": \"#\"}\n\t\t\t\t\t\t\t\t},\n\t\t\t\t\t\t\t\t\"required\": [\"back\"]\n\t\t\t\t\t\t\t}\n\t\t\t\t\t\t},\n\t\t\t\t\t\t\"required\": [\"level2\"]\n\t\t\t\t\t}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"level1\"]\n\t\t\t}","expect":"infinite recursion","unmarshal":false}
{"schema":"{\"type\":\"object\",\"properties\":{\"b\":{\"$ref\":\"#/properties/nope\"}}}","expect":"references must start with #/$defs/","unmarshal":false}
{"schema":"{\"type\":\"object\",\"properties\":{\"a\":{\"type\":\"string\"},\"b\":{\"$ref\":\"#/properties/a/type\"}}}","expect":"references must start with #/$defs/","unmarshal":false}
+{"schema": "{\"$defs\":{\"XXX\":{\"properties\":{\"next\":{\"$ref\":\"#/$defs/XXX\"}},\"required\":[\"next\"],\"type\":\"object\"}},\"properties\":{\"node\":{\"$ref\":\"#/$defs/XXX\"}},\"type\":\"object\"}", "expect": "detected infinite recursion without termination condition", "unmarshal": false}
+{"schema": "{\"type\":[\"object\"],\"properties\":{\"root\":{\"$ref\":\"#/$defs/Level1\"}},\"required\":[\"root\"],\"$defs\":{\"Level1\":{\"type\":[\"object\"],\"properties\":{\"level2\":{\"$ref\":\"#/$defs/Level2\"}},\"required\":[\"level2\"]},\"Level2\":{\"type\":[\"array\"],\"items\":{\"type\":[\"object\"],\"properties\":{\"level3\":{\"$ref\":\"#/$defs/Level3\"}},\"required\":[\"level3\"]},\"minItems\":1},\"Level3\":{\"type\":[\"object\"],\"properties\":{\"level1\":{\"$ref\":\"#/$defs/Level1\"}},\"required\":[\"level1\"]}}}", "expect": "detected infinite recursion without termination condition", "unmarshal": false}
+{"schema": "{\"type\":\"object\",\"properties\":{\"root\":{\"$ref\":\"#/$defs/A\"}},\"required\":[\"root\"],\"$defs\":{\"A\":{\"type\":\"object\",\"properties\":{\"b\":{\"$ref\":\"#/$defs/B\"}},\"required\":[\"b\"]},\"B\":{\"type\":\"object\",\"properties\":{\"a\":{\"$ref\":\"#/$defs/A\"}},\"required\":[\"a\"]}}}", "expect": "detected infinite recursion without termination condition", "unmarshal": false}
diff --git a/testdata/validator_cases/TestReferences/valid.jsonl b/testdata/validator_cases/TestReferences/valid.jsonl
index 0068a56..3c323c2 100644
--- a/testdata/validator_cases/TestReferences/valid.jsonl
+++ b/testdata/validator_cases/TestReferences/valid.jsonl
@@ -4,7 +4,6 @@
{"$defs":{"a":{"properties":{"y":{"$ref":"#/$defs/b"}},"type":"object"},"b":{"properties":{"z":{"$ref":"#/$defs/a"}},"type":"object"}},"properties":{"x":{"$ref":"#/$defs/a"}},"type":"object"}
{"$defs":{"Node":{"anyOf":[{"type":"string"},{"properties":{"next":{"$ref":"#/$defs/Node"}},"required":["next"],"type":"object"}]}},"properties":{"value":{"anyOf":[{"type":"string"},{"properties":{"next":{"$ref":"#/$defs/Node"}},"required":["next"],"type":"object"}]}},"type":"object"}
{"$defs":{"Node":{"properties":{"next":{"$ref":"#/$defs/Node"}},"type":"object"}},"properties":{"node":{"$ref":"#/$defs/Node"}},"required":["node"],"type":"object"}
-{"$defs":{"XXX":{"properties":{"next":{"$ref":"#/$defs/XXX"}},"required":["next"],"type":"object"}},"properties":{"node":{"$ref":"#/$defs/XXX"}},"type":"object"}
{"$defs":{"Node":{"properties":{"next":{"$ref":"#/$defs/Node"}},"type":"object"}},"properties":{"xxx":{"$ref":"#/$defs/Node"}},"type":"object"}
{"additionalProperties":false,"properties":{"attributes":{"description":"Arbitrary attributes for the UI component, suitable for any element","items":{"additionalProperties":false,"properties":{"name":{"description":"The name of the attribute, for example onClick or className","type":"string"},"value":{"description":"The value of the attribute","type":"string"}},"required":["name","value"],"type":"object"},"type":"array"},"children":{"description":"Nested UI components","items":{"$ref":"#"},"type":"array"},"label":{"description":"The label of the UI component, used for buttons or form fields","type":"string"},"type_info":{"description":"The type of the UI component","enum":["div","button","header","section","field","form"],"type":"string"}},"required":["type_info","label","children","attributes"],"type":"object"}
{"$defs":{"Def_0":{"description":"Schema structure data definition","items":{"anyOf":[{"items":{"anyOf":[{"type":"null"},{"type":"object"},{"type":"string"}]},"type":"array"},{"properties":{"lmGuD":{"type":"object"}},"required":["lmGuD"],"type":"object"}]},"type":"array"}},"description":"Example structure schema property","properties":{"JOvrn":{"$ref":"#/$defs/Def_0"},"riiIj":{"properties":{"MEaFc":{"items":{"type":"boolean"},"type":"array"}},"required":["MEaFc"],"type":"object"},"vXhVI":{"additionalProperties":false,"properties":{"mFlXV":{"properties":{"xWcek":{"type":"null"}},"required":["xWcek"],"type":"object"}},"required":["mFlXV"],"type":"object"}},"required":["riiIj"],"type":"object"}
@@ -21,3 +20,6 @@
{"properties":{"current_step":{"description":"step","minLength":1,"type":"string"},"final_summary":{"$ref":"#/properties/current_step","description":"summary"}},"type":"object"}
{"properties":{"list":{"items":{"type":"string"},"type":"array"},"one":{"$ref":"#/properties/list/items"}},"type":"object"}
{"$defs":{"S":{"type":"string"}},"properties":{"a":{"$ref":"#/$defs/S"},"b":{"type":"integer"},"c":{"$ref":"#/properties/b"}},"type":"object"}
+{"type":["object"],"properties":{"root":{"$ref":"#/$defs/Level1"}},"required":["root"],"$defs":{"Level1":{"type":["object"],"properties":{"level2":{"$ref":"#/$defs/Level2"}},"required":["level2"]},"Level2":{"type":["array"],"items":{"type":["object"],"properties":{"level3":{"$ref":"#/$defs/Level3"}},"required":["level3"]}},"Level3":{"type":["object"],"properties":{"level1":{"$ref":"#/$defs/Level1"}},"required":["level1"]}}}
+{"type":["object"],"properties":{"data":{"$ref":"#/$defs/InfiniteArray"}},"required":["data"],"$defs":{"InfiniteArray":{"type":["array"],"items":{"type":["object"],"properties":{"element":{"$ref":"#/$defs/InfiniteArray"}},"required":["element"]}}}}
+{"type":"object","properties":{"name":{"type":"string"},"children":{"type":"array","items":{"$ref":"#"}}},"required":["children"]}
diff --git a/testdata/validator_cases/TestRequired/invalid.jsonl b/testdata/validator_cases/TestRequired/invalid.jsonl
index ba979b2..8674a44 100644
--- a/testdata/validator_cases/TestRequired/invalid.jsonl
+++ b/testdata/validator_cases/TestRequired/invalid.jsonl
@@ -1,6 +1,6 @@
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"name\": {\"type\": \"string\"}\n\t\t\t\t},\n\t\t\t\t\"required\": \"name\"\n\t\t\t}","expect":"required must be an array","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"name\": {\"type\": \"string\"}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"name\", 123]\n\t\t\t}","expect":"items in required array must be strings","unmarshal":false}
-{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"name\": {\"type\": \"string\"}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"name\", \"\"]\n\t\t\t}","expect":"property names in required array cannot be empty","unmarshal":false}
+{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"name\": {\"type\": \"string\"}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"name\", \"\"]\n\t\t\t}","expect":"required property '' is not defined in properties","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"properties\": {\n\t\t\t\t\t\"name\": {\"type\": \"string\"}\n\t\t\t\t},\n\t\t\t\t\"required\": [\"age\"]\n\t\t\t}","expect":"required property 'age' is not defined in properties","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"string\",\n\t\t\t\t\"required\": [\"value\"]\n\t\t\t}","expect":"invalid keywords: required","unmarshal":false}
{"schema":"{\n\t\t\t\t\"type\": \"object\",\n\t\t\t\t\"required\": [\"name\"]\n\t\t\t}","expect":"'properties' keyword is missing","unmarshal":false}
diff --git a/testdata/validator_cases/TestRequired/valid.jsonl b/testdata/validator_cases/TestRequired/valid.jsonl
index 4ccc642..dcd8852 100644
--- a/testdata/validator_cases/TestRequired/valid.jsonl
+++ b/testdata/validator_cases/TestRequired/valid.jsonl
@@ -3,3 +3,4 @@
{"properties":{"age":{"type":"integer"},"email":{"type":"string"},"name":{"type":"string"}},"required":["name","age","email"],"type":"object"}
{"properties":{"user":{"properties":{"id":{"type":"string"},"info":{"type":"string"}},"required":["id"],"type":"object"}},"type":"object"}
{"$defs":{"FlightInfo":{"properties":{"date":{"description":"The date for the flight in the format 'YYYY-MM-DD', such as '2024-05-01'.","title":"Date","type":"string"},"flight_number":{"description":"Flight number, such as 'HAT001'.","title":"Flight Number","type":"string"}},"required":["flight_number","date"],"title":"FlightInfo","type":"object"},"Passenger":{"properties":{"dob":{"description":"Date of birth in YYYY-MM-DD format","title":"Dob","type":"string"},"first_name":{"description":"Passenger's first name","title":"First Name","type":"string"},"last_name":{"description":"Passenger's last name","title":"Last Name","type":"string"}},"required":["first_name","last_name","dob"],"title":"Passenger","type":"object"},"Payment":{"properties":{"amount":{"description":"Payment amount in dollars","title":"Amount","type":"integer"},"payment_id":{"description":"Unique identifier for the payment","title":"Payment Id","type":"string"}},"required":["payment_id","amount"],"title":"Payment","type":"object"}},"properties":{"cabin":{"description":"The cabin class such as 'basic_economy', 'economy', or 'business'.","enum":["business","economy","basic_economy"],"title":"Cabin","type":"string"},"destination":{"description":"The IATA code for the destination city such as 'JFK'.","title":"Destination","type":"string"},"flight_type":{"description":"The type of flight such as 'one_way' or 'round_trip'.","enum":["round_trip","one_way"],"title":"Flight Type","type":"string"},"flights":{"description":"An array of objects containing details about each piece of flight.","items":{"anyOf":[{"$ref":"#/$defs/FlightInfo"},{"type":"object"}]},"title":"Flights","type":"array"},"insurance":{"description":"Whether the reservation has insurance.","enum":["yes","no"],"title":"Insurance","type":"string"},"nonfree_baggages":{"description":"The number of non-free baggage items to book the reservation.","title":"Nonfree Baggages","type":"integer"},"origin":{"description":"The IATA code for the origin city such as 'SFO'.","title":"Origin","type":"string"},"passengers":{"description":"An array of objects containing details about each passenger.","items":{"anyOf":[{"$ref":"#/$defs/Passenger"},{"type":"object"}]},"title":"Passengers","type":"array"},"payment_methods":{"description":"An array of objects containing details about each payment method.","items":{"anyOf":[{"$ref":"#/$defs/Payment"},{"type":"object"}]},"title":"Payment Methods","type":"array"},"total_baggages":{"description":"The total number of baggage items to book the reservation.","title":"Total Baggages","type":"integer"},"user_id":{"description":"The ID of the user to book the reservation such as 'sara_doe_496'.","title":"User Id","type":"string"}},"required":["user_id","origin","destination","flight_type","cabin","flights","passengers","payment_methods","total_baggages","nonfree_baggages","insurance"],"title":"parameters","type":"object"}
+{"properties":{"":{"type":"string"},"name":{"type":"string"}},"required":["name",""],"type":"object"}
diff --git a/utils.go b/utils.go
index 358afa3..e4bd6e8 100644
--- a/utils.go
+++ b/utils.go
@@ -4,6 +4,7 @@ import (
"encoding/json"
"fmt"
"math"
+ "reflect"
"sort"
"strconv"
"strings"
@@ -86,6 +87,86 @@ func (p schemaPath) StringWithoutLast() schemaPath {
return newSchemaPathFromParts(p.Parts[:len(p.Parts)-1])
}
+// typeSet turns a type keyword value into a set of type names. Returns nil when
+// the value is missing or is not a usable type declaration.
+func typeSet(value any) map[string]struct{} {
+ switch t := value.(type) {
+ case string:
+ return map[string]struct{}{t: {}}
+ case SchemaList:
+ set := make(map[string]struct{}, len(t))
+ for _, item := range t {
+ if name, ok := item.(string); ok {
+ set[name] = struct{}{}
+ }
+ }
+ if len(set) == 0 {
+ return nil
+ }
+ return set
+ }
+ return nil
+}
+
+// typeSetsDisjoint reports whether no instance can satisfy both type sets at
+// once. integer and number are not disjoint: every integer is also a number, so
+// their intersection is integer. Unknown or absent sets are treated as
+// "no constraint" and therefore never disjoint.
+func typeSetsDisjoint(a, b map[string]struct{}) bool {
+ if len(a) == 0 || len(b) == 0 {
+ return false
+ }
+ for name := range a {
+ if _, ok := b[name]; ok {
+ return false
+ }
+ switch name {
+ case Integer:
+ if _, ok := b[Number]; ok {
+ return false
+ }
+ case Number:
+ if _, ok := b[Integer]; ok {
+ return false
+ }
+ }
+ }
+ return true
+}
+
+// enumValuesDisjoint reports whether two enum lists have no value in common.
+// Absent or empty lists impose no constraint and are never disjoint.
+func enumValuesDisjoint(a, b any) bool {
+ listA, okA := a.(SchemaList)
+ listB, okB := b.(SchemaList)
+ if !okA || !okB || len(listA) == 0 || len(listB) == 0 {
+ return false
+ }
+
+ for _, left := range listA {
+ for _, right := range listB {
+ if reflect.DeepEqual(left, right) {
+ return false
+ }
+ }
+ }
+ return true
+}
+
+// unsatisfiableOverlap names a keyword that the two schemas both constrain in
+// ways that cannot hold at once, or "" when every shared keyword still admits
+// some instance. Numeric bounds are absent on purpose: conjoining two minLength
+// values just keeps the larger one, which is always satisfiable.
+func unsatisfiableOverlap(parent, refSchema SchemaDict) string {
+ if typeSetsDisjoint(typeSet(parent[Type]), typeSet(refSchema[Type])) {
+ return Type
+ }
+ if enumValuesDisjoint(parent[Enum], refSchema[Enum]) {
+ return Enum
+ }
+ return ""
+}
+
// validateUtils provides utility functions for schema validation
type validateUtils struct{}
diff --git a/utils_test.go b/utils_test.go
index 2d95cf5..678fb53 100644
--- a/utils_test.go
+++ b/utils_test.go
@@ -399,14 +399,15 @@ func TestTraverseAndCheckRefs(t *testing.T) {
must.Contains(strings.ToLower(err.Error()), "detected infinite recursion")
if validator.config.IsUltra() || validator.config.IsTest() {
- // Test reference with conflicting keywords
+ // number alongside a reference to a string: the two types have no common
+ // instance, so this is unsatisfiable rather than a mergeable duplicate
schema = SchemaDict{
"$ref": "#/$defs/simpleType",
"type": "number",
}
err = validator.TraverseAndCheckRefs(schema, true, nil, newSchemaPath("properties.conflicting"))
must.Error(err)
- must.Contains(strings.ToLower(err.Error()), "conflicting keywords")
+ must.Contains(strings.ToLower(err.Error()), "intersection is empty")
}
// Test traversal of complex nested structure
@@ -537,16 +538,29 @@ func TestCheckRefContext(t *testing.T) {
must.NoError(err)
if validator.config.IsUltra() || validator.config.IsTest() {
- // Test with direct keyword conflict
+ // Contradictory types cannot be merged at all
parent = SchemaDict{
"$ref": "#/$defs/type",
- "type": "number", // Conflicts with refSchema
+ "type": "number", // no instance is both a number and a string
}
refSchema = SchemaDict{
"type": "string",
}
err = validator.CheckRefContext(parent, refSchema, newSchemaPath(""))
must.Error(err)
+ must.Contains(strings.ToLower(err.Error()), "intersection is empty")
+
+ // A duplicate that could be merged is reported as a keyword conflict
+ parent = SchemaDict{
+ "$ref": "#/$defs/type",
+ "minLength": 20.0,
+ }
+ refSchema = SchemaDict{
+ "type": "string",
+ "minLength": 10.0,
+ }
+ err = validator.CheckRefContext(parent, refSchema, newSchemaPath(""))
+ must.Error(err)
must.Contains(strings.ToLower(err.Error()), "conflicting keywords")
// Test type + anyOf conflict
@@ -618,26 +632,38 @@ func TestRefCommonKeywordConflictByValidateLevel(t *testing.T) {
must.Contains(strings.ToLower(err.Error()), "description")
})
- t.Run("lite still rejects structural keyword conflicts after ref expansion", func(t *testing.T) {
- must := require.New(t)
- schema := `{
- "type": "object",
- "properties": {
- "value": {
- "$ref": "#/$defs/ValueType",
- "type": "string"
- }
- },
- "$defs": {
- "ValueType": { "type": "number" }
+ // A contradiction inside one node is reported from strict upwards, the same way
+ // a lower bound above its upper bound is. Lite accepts the schema and leaves it
+ // to canonicalisation to drop the contradicting keyword.
+ unsatisfiableTypeBesideRef := `{
+ "type": "object",
+ "properties": {
+ "value": {
+ "$ref": "#/$defs/ValueType",
+ "type": "string"
}
- }`
+ },
+ "$defs": {
+ "ValueType": { "type": "number" }
+ }
+ }`
+
+ t.Run("lite allows an unsatisfiable type beside $ref", func(t *testing.T) {
+ must := require.New(t)
validator := newSchemaValidator(WithValidateLevel(ValidateLevelLite))
- err := validator.Validate(schema)
- must.Error(err)
- must.Contains(strings.ToLower(err.Error()), "type")
+ must.NoError(validator.Validate(unsatisfiableTypeBesideRef))
})
+ for _, level := range []ValidateLevel{ValidateLevelStrict, ValidateLevelUltra} {
+ t.Run(string(level)+" rejects an unsatisfiable type beside $ref", func(t *testing.T) {
+ must := require.New(t)
+ validator := newSchemaValidator(WithValidateLevel(level))
+ err := validator.Validate(unsatisfiableTypeBesideRef)
+ must.Error(err)
+ must.Contains(strings.ToLower(err.Error()), "intersection is empty")
+ })
+ }
+
schemaWithAnyOfParentDescriptionConflict := `{
"type": "object",
"properties": {
@@ -771,12 +797,20 @@ func TestRefCommonKeywordConflictByValidateLevel(t *testing.T) {
}
}`
- t.Run("lite still rejects structural keyword conflict in anyOf with parent", func(t *testing.T) {
+ // Constraining an instance both directly and inside a branch is a legal
+ // conjunction, so lite accepts it and leaves canonicalisation to push the outer
+ // copy into the branches.
+ t.Run("lite allows a structural keyword stated both beside and inside anyOf", func(t *testing.T) {
must := require.New(t)
validator := newSchemaValidator(WithValidateLevel(ValidateLevelLite))
+ must.NoError(validator.Validate(schemaWithAnyOfParentMinLengthConflict))
+ })
+
+ t.Run("ultra reports it so that Canonical distributes it", func(t *testing.T) {
+ must := require.New(t)
+ validator := newSchemaValidator(WithValidateLevel(ValidateLevelUltra))
err := validator.Validate(schemaWithAnyOfParentMinLengthConflict)
must.Error(err)
- must.Contains(strings.ToLower(err.Error()), "conflicting keywords")
must.Contains(strings.ToLower(err.Error()), "minlength")
})
diff --git a/validator.go b/validator.go
index 2e6916c..fe54e3c 100644
--- a/validator.go
+++ b/validator.go
@@ -18,10 +18,21 @@ type schemaValidator struct {
validateItemsRange KeywordValidatorFunc
defDepths map[string]int
totalPropKeys int
+ terminationMemo map[string]bool
+ refWalkSteps int
utils *validateUtils
config SchemaValidatorConfig
}
+// maxRefWalkSteps caps the total number of nodes the reference walkers (defs
+// depth computation and reference termination checking) may visit during one
+// validation. Chains of definitions forming diamonds -- or diamonds closing a
+// long cycle -- cost 2^n walks without memoization; past this budget
+// validation fails closed instead of hanging on adversarial schemas. Honest
+// schemas stay orders of magnitude below it: memoized walks are linear in the
+// number of definitions.
+const maxRefWalkSteps = 500000
+
func newSchemaValidator(options ...SchemaValidatorOption) *schemaValidator {
config := DefaultValidatorConfig()
for _, option := range options {
@@ -64,6 +75,8 @@ func (v *schemaValidator) Reset() {
v.context = newValidationContext()
v.defDepths = make(map[string]int)
v.totalPropKeys = 0
+ v.terminationMemo = nil
+ v.refWalkSteps = 0
// won't reset config
}
@@ -109,7 +122,12 @@ func (v *schemaValidator) CheckAnyOfConflicts(schema SchemaDict, path schemaPath
var conflicts []string
for k := range branchKeywords {
if _, exists := outerKeywords[k]; exists {
- if CommonKeywords[k] && !(v.config.IsUltra() || v.config.IsTest()) {
+ // Constraining an instance both directly and inside a branch is legal
+ // under 2020-12: both apply and the effective constraint is their
+ // conjunction. Only the canonicalising levels report it, so that
+ // Canonical distributes the parent copy into the branches while looser
+ // levels accept the schema as written.
+ if !(v.config.IsUltra() || v.config.IsTest()) {
continue
}
conflicts = append(conflicts, k)
@@ -123,7 +141,7 @@ func (v *schemaValidator) CheckAnyOfConflicts(schema SchemaDict, path schemaPath
`conflicting keywords found in anyOf with parent: keywords (%s) are defined on the parent schema and inside anyOf; remove them from the parent or from anyOf branches`,
strings.Join(conflicts, ", "),
),
- path, simplifyFuncForKeywordConflicts(conflicts),
+ path, simplifyFuncForAnyOfParentConflicts(conflicts),
)
}
}
@@ -153,14 +171,12 @@ func (v *schemaValidator) TraverseSchema(schema SchemaDict, path schemaPath, cur
if len(unsupported) > 0 && (v.config.IsUltra() || v.config.IsTest()) {
sort.Strings(unsupported)
- simplifyFunc := SimplifyDefault
- for _, keyword := range unsupported {
- if keyword == "$schema" {
- simplifyFunc = SimplifyRemoveSchemaKeys([]string{"$schema"})
- break
- }
- }
- return currentDepth, v.context.RaiseErrorWithSimplify(fmt.Sprintf("unsupported keywords: %s", strings.Join(unsupported, ", ")), path, simplifyFunc)
+ // Drop just the unsupported keywords; the enforcer ignores keys it does not
+ // know, so keeping the rest of the schema is both safe and closer to intent.
+ return currentDepth, v.context.RaiseErrorWithSimplify(
+ fmt.Sprintf("unsupported keywords: %s", strings.Join(unsupported, ", ")),
+ path, SimplifyRemoveSchemaKeys(unsupported),
+ )
}
// Process $defs first
@@ -190,19 +206,52 @@ func (v *schemaValidator) TraverseSchema(schema SchemaDict, path schemaPath, cur
}
}
+ // A contradiction between $ref and its siblings is reported before any other
+ // $ref rule. The rules below delete siblings to canonicalise the node, and
+ // once a contradicting sibling is gone the remaining schema happily accepts
+ // what the original rejected.
+ //
+ // Strict and above reject it, matching how a lower bound above its upper bound
+ // is handled: both are contradictions inside one node, and neither used to
+ // stop lite. Looser levels accept the schema and rely on canonicalisation to
+ // drop the contradicting keyword.
+ if _, hasRef := schema[Ref]; hasRef && v.config.IsGreaterThanStrict() {
+ if keyword := v.refSiblingContradiction(schema, path); keyword != "" {
+ return currentDepth, v.context.RaiseErrorWithSimplify(
+ fmt.Sprintf(
+ "%s conflicts with the referenced schema: the intersection is empty, so no instance can satisfy both",
+ keyword,
+ ),
+ path, SimplifyDropContradictingRefSibling,
+ )
+ }
+ }
+
// Check type and anyOf/ref conflicts
if _, hasType := schema[Type]; hasType {
if _, hasAnyOf := schema[AnyOf]; hasAnyOf {
- return currentDepth, v.context.RaiseErrorWithSimplify(
- "when using anyOf, type should be defined in anyOf items instead of the parent schema",
- path, SimplifyRemoveType,
- )
+ // A type beside anyOf is legal under 2020-12: it applies on top of
+ // whichever branch matches. The enforcer's anyOf cannot carry a conjunct,
+ // so only the canonicalising levels report it and the type is pushed into
+ // the branches; looser levels accept the schema as written.
+ if v.config.IsUltra() || v.config.IsTest() {
+ return currentDepth, v.context.RaiseErrorWithSimplify(
+ "when using anyOf, type should be defined in anyOf items instead of the parent schema",
+ path, SimplifyDistributeAnyOfParent,
+ )
+ }
}
if _, hasRef := schema[Ref]; hasRef {
- return currentDepth, v.context.RaiseErrorWithSimplify(
- "when using $ref, type should be defined in the referenced schema instead of the parent schema",
- path, SimplifyRemoveType,
- )
+ // A compatible type next to $ref is legal under 2020-12: both assertions
+ // apply and the intersection is non-empty, which the check above already
+ // established. Only the canonicalising levels fold it away, so looser
+ // levels accept the schema as written.
+ if v.config.IsUltra() || v.config.IsTest() {
+ return currentDepth, v.context.RaiseErrorWithSimplify(
+ "when using $ref, type should be defined in the referenced schema instead of the parent schema",
+ path, SimplifyRemoveType,
+ )
+ }
}
if typeList, ok := schema[Type].(SchemaList); ok && len(typeList) > 1 {
@@ -403,12 +452,23 @@ func (v *schemaValidator) validateTypeAndKeywords(schema SchemaDict, path schema
}
}
+ // A lower bound above its upper bound admits no instance at all, so strict
+ // reports it too even though the keyword checks below stay ultra-only. A
+ // type this function does not recognise is left to ValidateType to report.
+ if len(types) >= 1 && v.config.IsStrict() {
+ if allowedKeywords, err := v.computeAllowedKeywordsForTypes(path, types); err == nil {
+ if err := v.validateRangeKeywordsForAllowedTypes(schema, path, allowedKeywords); err != nil {
+ return err
+ }
+ }
+ }
+
if len(types) >= 1 && (v.config.IsUltra() || v.config.IsTest()) {
// Check $defs and $id are only at top level
if !path.IsRoot() {
for k := range schema {
if TopLevelOnlyKeywords[k] {
- return v.context.RaiseErrorWithSimplify(fmt.Sprintf("keyword %s must be at root level", k), path, SimplifyDefault)
+ return v.context.RaiseErrorWithSimplify(fmt.Sprintf("keyword %s must be at root level", k), path, SimplifyRemoveSchemaKeys([]string{k}))
}
}
}
@@ -574,12 +634,35 @@ func (v *schemaValidator) Validate(schema any) error {
}
func (v *schemaValidator) CanonicalWithMaxAttempts(schema Schema, maxAttempts int) (string, error) {
- currentSchema := Schema(hoistLocalRefs(schema))
-
+ // Fold sibling constraints into their $ref target or anyOf branches before
+ // validating. Left to the retry loop these siblings would simply be deleted,
+ // which silently loosens the schema whenever the sibling was the stricter of
+ // the two. anyOf is distributed first because doing so can leave a constraint
+ // beside a branch's $ref, which is what the inlining pass then folds in.
+ inlined, droppedSiblings := inlineConflictingRefSiblings(
+ distributeAnyOfParentKeywords(hoistLocalRefs(schema)),
+ v.config.MaxSchemaSize,
+ )
+ currentSchema := Schema(inlined)
+
+ // Siblings the copy budget could not fold in were dropped in one pass rather
+ // than one retry at a time; surface exactly what was lost as the warning.
var rawErr error
+ if len(droppedSiblings) > 0 {
+ const maxReported = 5
+ shown := droppedSiblings
+ if len(shown) > maxReported {
+ shown = droppedSiblings[:maxReported]
+ }
+ rawErr = fmt.Errorf(
+ "inlining every $ref would exceed the schema size limit, so %d sibling constraint(s) were dropped instead: %s",
+ len(droppedSiblings), strings.Join(shown, ", "),
+ )
+ }
+
for i := 0; i < maxAttempts; i++ {
err := v.Validate(currentSchema)
- if i == 0 {
+ if i == 0 && err != nil {
rawErr = err
}
if err == nil {
@@ -623,7 +706,11 @@ func (v *schemaValidator) validateSchemaDict(schema SchemaDict) error {
}
// Precompute defs depth
- v.defDepths = v.CalculateDefDepths()
+ defDepths, err := v.CalculateDefDepths()
+ if err != nil {
+ return err
+ }
+ v.defDepths = defDepths
// Verify schema
maxDepth, err := v.TraverseSchema(schema, rootSchemaPath, 0)
@@ -639,6 +726,24 @@ func (v *schemaValidator) validateSchemaDict(schema SchemaDict) error {
return v.PostValidateRefs()
}
+// refSiblingContradiction names a keyword that the node and the schema it
+// references constrain in ways that cannot both hold, or "" when they can. A
+// reference that cannot be resolved is not a contradiction; PostValidateRefs
+// reports that separately.
+func (v *schemaValidator) refSiblingContradiction(schema SchemaDict, path schemaPath) string {
+ refStr, ok := schema[Ref].(string)
+ if !ok {
+ return ""
+ }
+
+ target, err := v.utils.ResolveRef(v.context.SchemaRoot, refStr, v.context, path)
+ if err != nil || target == nil {
+ return ""
+ }
+
+ return unsatisfiableOverlap(schema, target)
+}
+
// PostValidateRefs validates all references after schema traversal
func (v *schemaValidator) PostValidateRefs() error {
// Verify all ref paths exist
@@ -648,15 +753,10 @@ func (v *schemaValidator) PostValidateRefs() error {
}
}
- // first check root schema whether it can terminate
- needCheckTermination := true
- if terminates, err := v.CheckRefTermination(v.context.SchemaRoot, make(map[string]struct{}), rootSchemaPath); err == nil {
- if terminates {
- needCheckTermination = false
- }
- }
- // Traverse and check all references
- return v.TraverseAndCheckRefs(v.context.SchemaRoot, needCheckTermination, nil, rootSchemaPath)
+ // Every $ref has to be able to terminate. Short-circuiting on the root was not
+ // enough: the root normally terminates because its own properties are optional,
+ // which masked non-terminating definitions nested inside $defs.
+ return v.TraverseAndCheckRefs(v.context.SchemaRoot, true, nil, rootSchemaPath)
}
// TraverseAndCheckRefs traverses the schema and checks all references
@@ -749,26 +849,51 @@ func (v *schemaValidator) TraverseAndCheckRefs(schema SchemaDict, needCheckTermi
return nil
}
-func (v *schemaValidator) CalculateDefDepths() map[string]int {
+// CalculateDefDepths computes the nesting depth each definition expands to.
+// Like the termination check it memoizes expanded definitions -- keyed by the
+// same purity rule, a depth computed without running into the path stack -- and
+// gives up with an error past the shared step budget.
+func (v *schemaValidator) CalculateDefDepths() (map[string]int, error) {
defDepths := make(map[string]int)
+ memo := make(map[string]int)
+
+ var calculateDepthsRecursive func(schema SchemaDict, currentPath string, visitedRefs map[string]struct{}) (int, bool, error)
+ calculateDepthsRecursive = func(schema SchemaDict, currentPath string, visitedRefs map[string]struct{}) (int, bool, error) {
+ v.refWalkSteps++
+ if v.refWalkSteps > maxRefWalkSteps {
+ return 0, false, v.context.RaiseError("reference graph is too complex to validate within the step budget", rootSchemaPath)
+ }
- var calculateDepthsRecursive func(schema SchemaDict, currentPath string, visitedRefs map[string]struct{}) int
- calculateDepthsRecursive = func(schema SchemaDict, currentPath string, visitedRefs map[string]struct{}) int {
// The depth of basic type or empty schema is 0
if len(schema) == 0 {
- return 0
+ return 0, false, nil
}
// Record the depth of current path
defDepths[currentPath] = 0 // Initial depth is 0
maxDepth := 0
+ cut := false
if ref, ok := schema[Ref].(string); ok {
- if _, exists := visitedRefs[ref]; !exists {
+ if _, exists := visitedRefs[ref]; exists {
+ // Cycle guard: the ref contributes no depth, and the result may
+ // depend on the entry stack, so it must not be cached.
+ cut = true
+ } else if depth, done := memo[ref]; done {
+ maxDepth = depth
+ } else {
visitedRefs[ref] = struct{}{}
resolved, err := v.utils.ResolveRef(v.context.SchemaRoot, ref, v.context, rootSchemaPath)
if err == nil && resolved != nil {
- maxDepth = calculateDepthsRecursive(resolved, ref, visitedRefs)
+ depth, refCut, err := calculateDepthsRecursive(resolved, ref, visitedRefs)
+ if err != nil {
+ return 0, false, err
+ }
+ if !refCut {
+ memo[ref] = depth
+ }
+ cut = cut || refCut
+ maxDepth = depth
}
}
}
@@ -788,7 +913,11 @@ func (v *schemaValidator) CalculateDefDepths() map[string]int {
propVisited[k] = v
}
- subDepth := calculateDepthsRecursive(propSchemaObj, propPath, propVisited)
+ subDepth, propCut, err := calculateDepthsRecursive(propSchemaObj, propPath, propVisited)
+ if err != nil {
+ return 0, false, err
+ }
+ cut = cut || propCut
if 1+subDepth > propsDepth {
propsDepth = 1 + subDepth
}
@@ -812,7 +941,11 @@ func (v *schemaValidator) CalculateDefDepths() map[string]int {
branchVisited[k] = v
}
- subDepth := calculateDepthsRecursive(subSchemaObj, subPath, branchVisited)
+ subDepth, branchCut, err := calculateDepthsRecursive(subSchemaObj, subPath, branchVisited)
+ if err != nil {
+ return 0, false, err
+ }
+ cut = cut || branchCut
if subDepth > maxDepth {
maxDepth = subDepth
}
@@ -828,7 +961,11 @@ func (v *schemaValidator) CalculateDefDepths() map[string]int {
addPropsVisited[k] = v
}
- subDepth := calculateDepthsRecursive(addProps, addPropsPath, addPropsVisited)
+ subDepth, addPropsCut, err := calculateDepthsRecursive(addProps, addPropsPath, addPropsVisited)
+ if err != nil {
+ return 0, false, err
+ }
+ cut = cut || addPropsCut
if subDepth > maxDepth {
maxDepth = subDepth
}
@@ -836,7 +973,7 @@ func (v *schemaValidator) CalculateDefDepths() map[string]int {
// Update the final depth of current path
defDepths[currentPath] = maxDepth
- return maxDepth
+ return maxDepth, cut, nil
}
// Traverse from $defs
@@ -847,28 +984,55 @@ func (v *schemaValidator) CalculateDefDepths() map[string]int {
continue
}
basePath := fmt.Sprintf("#/$defs/%s", defName)
- calculateDepthsRecursive(defSchemaObj, basePath, make(map[string]struct{}))
+ if _, _, err := calculateDepthsRecursive(defSchemaObj, basePath, make(map[string]struct{})); err != nil {
+ return nil, err
+ }
}
}
- return defDepths
+ return defDepths, nil
}
// CheckRefTermination checks if a reference can be terminated
func (v *schemaValidator) CheckRefTermination(schema SchemaDict, visitedRefs map[string]struct{}, path schemaPath) (bool, error) {
+ // The memo is shared across every call of one validation, so a definition
+ // is expanded once no matter how many use sites point at it.
+ if v.terminationMemo == nil {
+ v.terminationMemo = make(map[string]bool)
+ }
+ terminates, _, err := v.checkRefTermination(schema, visitedRefs, v.terminationMemo, path)
+ return terminates, err
+}
+
+// checkRefTermination is the recursive body of CheckRefTermination. The memo
+// caches the verdict of each fully expanded $ref, so sibling properties that
+// point at the same definition do not each re-walk its whole subgraph --
+// without it a chain of definitions referenced twice per level costs 2^n walks.
+//
+// A verdict is cached only when computing it never ran into a reference already
+// on the path stack: such a verdict is a property of the definition alone and
+// holds for every later entry point. A walk that did run into the stack may
+// have been cut short by entry-specific state, so its verdict is used but not
+// cached. The second return value reports whether that happened.
+func (v *schemaValidator) checkRefTermination(schema SchemaDict, visitedRefs map[string]struct{}, memo map[string]bool, path schemaPath) (bool, bool, error) {
+ v.refWalkSteps++
+ if v.refWalkSteps > maxRefWalkSteps {
+ return false, false, v.context.RaiseError("reference graph is too complex to validate within the step budget", path)
+ }
+
// Non-object/array basic types can terminate
var checkType string
if typeVal, ok := schema[Type]; ok {
switch t := typeVal.(type) {
case string:
if t != Object && t != Array {
- return true, nil
+ return true, false, nil
}
checkType = t
case SchemaList:
for _, typ := range t {
if typeStr, ok := typ.(string); ok && typeStr != Object && typeStr != Array {
- return true, nil
+ return true, false, nil
}
// should be object or array and len(t) == 1/2
@@ -877,26 +1041,36 @@ func (v *schemaValidator) CheckRefTermination(schema SchemaDict, visitedRefs map
}
}
+ cut := false
+
// Array items if empty can terminate
if checkType == Array {
+ // Without a positive minItems the empty array is a valid instance, so the
+ // array terminates no matter what items demands. Only a non-empty lower
+ // bound forces us to prove that items itself can terminate.
+ if minItems, ok := schema[MinItems].(float64); !ok || minItems <= 0 {
+ return true, false, nil
+ }
+
items, exists := schema[Items]
if !exists || items == nil {
- return true, nil
+ return true, false, nil
}
if itemsDict, ok := items.(SchemaDict); ok && len(itemsDict) == 0 {
- return true, nil
+ return true, false, nil
}
// check array items whether it can terminate
if itemsDict, ok := items.(SchemaDict); ok {
- terminates, err := v.CheckRefTermination(itemsDict, visitedRefs, path.Append(Items))
+ terminates, itemsCut, err := v.checkRefTermination(itemsDict, visitedRefs, memo, path.Append(Items))
if err != nil {
- return false, err
+ return false, false, err
}
if terminates {
- return true, nil
+ return true, itemsCut, nil
}
+ cut = cut || itemsCut
}
}
@@ -905,17 +1079,17 @@ func (v *schemaValidator) CheckRefTermination(schema SchemaDict, visitedRefs map
required, exists := schema[Required]
// required if empty can terminate
if !exists || required == nil {
- return true, nil
+ return true, false, nil
}
if requiredList, ok := required.(SchemaList); ok && len(requiredList) == 0 {
- return true, nil
+ return true, false, nil
}
// if properties is empty, it can terminate
props, hasProps := schema[Properties].(SchemaDict)
if !hasProps || len(props) == 0 {
- return true, nil
+ return true, false, nil
}
// iterate properties
@@ -926,10 +1100,13 @@ func (v *schemaValidator) CheckRefTermination(schema SchemaDict, visitedRefs map
}
}
+ // Every required property has to be present in a valid instance, so the
+ // object only terminates when all of them do. A single non-terminating
+ // required property makes the object unsatisfiable by any finite document.
for propName, propSchema := range props {
propSchemaObj, ok := propSchema.(SchemaDict)
if !ok {
- return false, v.context.RaiseErrorWithSimplify("property schema must be an object", path.Append(propName), SimplifyRemoveProperties)
+ return false, false, v.context.RaiseErrorWithSimplify("property schema must be an object", path.Append(propName), SimplifyRemoveProperties)
}
// only check required properties
@@ -937,26 +1114,38 @@ func (v *schemaValidator) CheckRefTermination(schema SchemaDict, visitedRefs map
continue
}
- if terminates, err := v.CheckRefTermination(propSchemaObj, visitedRefs, path.Append(Properties, propName)); err != nil {
- return false, err
- } else if terminates {
- return true, nil
+ // Each property walks its own visited set; two sibling properties
+ // pointing at the same $ref is not a cycle.
+ propRefs := make(map[string]struct{})
+ for k := range visitedRefs {
+ propRefs[k] = struct{}{}
}
+
+ terminates, propCut, err := v.checkRefTermination(propSchemaObj, propRefs, memo, path.Append(Properties, propName))
+ if err != nil {
+ return false, false, err
+ }
+ if !terminates {
+ return false, propCut, nil
+ }
+ cut = cut || propCut
}
+
+ return true, cut, nil
}
// Check anyOf branches
if anyOf, ok := schema[AnyOf].(SchemaList); ok {
// if anyOf is empty, it can terminate
if len(anyOf) == 0 {
- return true, nil
+ return true, false, nil
}
allRefs := make(map[string]struct{})
for i, item := range anyOf {
itemSchema, ok := item.(SchemaDict)
if !ok {
- return false, v.context.RaiseErrorWithSimplify("schema in anyOf must be an object", path.Append(AnyOf), SimplifyRemoveAnyOf)
+ return false, false, v.context.RaiseErrorWithSimplify("schema in anyOf must be an object", path.Append(AnyOf), SimplifyRemoveAnyOf)
}
// Create a new copy of visited refs for each branch
@@ -965,15 +1154,18 @@ func (v *schemaValidator) CheckRefTermination(schema SchemaDict, visitedRefs map
branchRefs[k] = struct{}{}
}
- if terminates, err := v.CheckRefTermination(itemSchema, branchRefs, path.Append(AnyOf, strconv.Itoa(i))); err != nil {
- return false, err
- } else if terminates {
+ terminates, branchCut, err := v.checkRefTermination(itemSchema, branchRefs, memo, path.Append(AnyOf, strconv.Itoa(i)))
+ if err != nil {
+ return false, false, err
+ }
+ if terminates {
// Update visited refs with branch refs
for k := range branchRefs {
visitedRefs[k] = struct{}{}
}
- return true, nil
+ return true, branchCut, nil
}
+ cut = cut || branchCut
// Collect all refs from this branch
for k := range branchRefs {
@@ -990,23 +1182,34 @@ func (v *schemaValidator) CheckRefTermination(schema SchemaDict, visitedRefs map
// Check ref
if ref, ok := schema[Ref].(string); ok {
if _, exists := visitedRefs[ref]; exists {
- return false, nil
+ return false, true, nil
+ }
+
+ if result, done := memo[ref]; done {
+ return result, false, nil
}
visitedRefs[ref] = struct{}{}
target, err := v.utils.ResolveRef(v.context.SchemaRoot, ref, v.context, path)
if err != nil {
- return false, v.context.RaiseError(fmt.Sprintf("invalid $ref path: %s", ref), path)
+ return false, false, v.context.RaiseError(fmt.Sprintf("invalid $ref path: %s", ref), path)
}
- return v.CheckRefTermination(target, visitedRefs, path.Append(Ref))
+ result, targetCut, err := v.checkRefTermination(target, visitedRefs, memo, path.Append(Ref))
+ if err != nil {
+ return false, false, err
+ }
+ if !targetCut {
+ memo[ref] = result
+ }
+ return result, targetCut, nil
}
if len(schema) == 0 {
- return true, nil
+ return true, false, nil
}
- return false, nil
+ return false, cut, nil
}
// ExpandRef expands a reference
@@ -1136,10 +1339,27 @@ func (v *schemaValidator) CheckRefContext(parent SchemaDict, refSchema SchemaDic
}
}
+ // Contradictory overlaps come first: their conjunction admits no instance at
+ // all, so no amount of merging produces a usable schema. Only strict and above
+ // reject them; see TraverseSchema for why looser levels let them through.
+ if keyword := unsatisfiableOverlap(parent, refSchema); keyword != "" && v.config.IsGreaterThanStrict() {
+ return v.context.RaiseErrorWithSimplify(
+ fmt.Sprintf(
+ "%s conflicts with the referenced schema: the intersection is empty, so no instance can satisfy both",
+ keyword,
+ ),
+ path, SimplifyDropContradictingRefSibling,
+ )
+ }
+
var conflicts []string
for k := range refSchema {
if _, exists := parentKeywords[k]; exists {
- if CommonKeywords[k] && !(v.config.IsUltra() || v.config.IsTest()) {
+ // Repeating a keyword next to $ref is legal under 2020-12: both
+ // assertions apply and the effective constraint is their conjunction.
+ // Only the canonicalising levels report it, so that Canonical folds the
+ // duplicate away while looser levels accept the schema as written.
+ if !(v.config.IsUltra() || v.config.IsTest()) {
continue
}
conflicts = append(conflicts, k)
diff --git a/validator_test.go b/validator_test.go
index de68c59..0925864 100644
--- a/validator_test.go
+++ b/validator_test.go
@@ -733,3 +733,530 @@ func TestLiteAllowsReservedPropertyNamesInProperties(t *testing.T) {
must.Error(err, "ultra should reject reserved property names")
must.Contains(strings.ToLower(err.Error()), "reserved")
}
+
+// Termination is about whether a *finite* instance exists, not about whether the
+// schema mentions itself. A cycle only makes the schema unsatisfiable when every
+// hop is mandatory: a required object property always forces one more level,
+// while an array without a positive minItems can stop at the empty array.
+//
+// This used to be masked twice over: PostValidateRefs skipped the whole check
+// whenever the root happened to terminate, and CheckRefTermination treated the
+// required properties of an object as "any one of them terminates" instead of
+// "all of them must".
+func TestRefTerminationDistinguishesFiniteFromInfiniteRecursion(t *testing.T) {
+ cases := []struct {
+ name string
+ // witness is a JSON instance proving satisfiability, or why none exists
+ witness string
+ schema string
+ reject bool
+ }{
+ {
+ name: "optional self reference stops immediately",
+ witness: `{"node":{}}`,
+ schema: `{"type":"object","properties":{"node":{"$ref":"#/$defs/N"}},"$defs":{"N":{"type":"object","properties":{"next":{"$ref":"#/$defs/N"}}}}}`,
+ },
+ {
+ name: "required array self reference stops at the empty array",
+ witness: `{"children":[]}`,
+ schema: `{"type":"object","properties":{"children":{"type":"array","items":{"$ref":"#"}}},"required":["children"]}`,
+ },
+ {
+ name: "mutual recursion through an unbounded array stops at the empty array",
+ witness: `{"root":{"level2":[]}}`,
+ schema: `{"type":"object","properties":{"root":{"$ref":"#/$defs/L1"}},"required":["root"],"$defs":{"L1":{"type":"object","properties":{"level2":{"$ref":"#/$defs/L2"}},"required":["level2"]},"L2":{"type":"array","items":{"$ref":"#/$defs/L1"}}}}`,
+ },
+ {
+ name: "required self reference never bottoms out",
+ witness: "none: every instance needs one more 'next'",
+ schema: `{"type":"object","properties":{"node":{"$ref":"#/$defs/N"}},"$defs":{"N":{"type":"object","properties":{"next":{"$ref":"#/$defs/N"}},"required":["next"]}}}`,
+ reject: true,
+ },
+ {
+ name: "mutual recursion where every hop is required",
+ witness: "none: A needs B, B needs A",
+ schema: `{"type":"object","properties":{"root":{"$ref":"#/$defs/A"}},"required":["root"],"$defs":{"A":{"type":"object","properties":{"b":{"$ref":"#/$defs/B"}},"required":["b"]},"B":{"type":"object","properties":{"a":{"$ref":"#/$defs/A"}},"required":["a"]}}}`,
+ reject: true,
+ },
+ {
+ name: "minItems forces the array cycle to continue",
+ witness: "none: the array can never be empty",
+ schema: `{"type":"object","properties":{"root":{"$ref":"#/$defs/L1"}},"required":["root"],"$defs":{"L1":{"type":"object","properties":{"level2":{"$ref":"#/$defs/L2"}},"required":["level2"]},"L2":{"type":"array","minItems":1,"items":{"$ref":"#/$defs/L1"}}}}`,
+ reject: true,
+ },
+ {
+ name: "non-terminating branch hidden behind a terminating sibling",
+ witness: "none: 'loop' is required alongside the harmless 'label'",
+ schema: `{"type":"object","properties":{"label":{"type":"string"},"loop":{"$ref":"#/$defs/N"}},"required":["label","loop"],"$defs":{"N":{"type":"object","properties":{"next":{"$ref":"#/$defs/N"}},"required":["next"]}}}`,
+ reject: true,
+ },
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ err := newSchemaValidator(WithValidateLevel(ValidateLevelLite)).Validate(tc.schema)
+ if tc.reject {
+ if err == nil {
+ t.Fatalf("expected rejection (%s)", tc.witness)
+ }
+ if !strings.Contains(err.Error(), "infinite recursion") {
+ t.Fatalf("expected an infinite recursion error, got %v", err)
+ }
+ return
+ }
+ if err != nil {
+ t.Fatalf("expected schema to pass, %s is a valid instance: %v", tc.witness, err)
+ }
+ })
+ }
+}
+
+// A lower bound above its upper bound cannot be satisfied by any instance. lite
+// keeps accepting such schemas so that existing callers do not start failing,
+// but strict and above reject them, and Canonical degrades the offending
+// subschema to {} rather than deleting the bounds -- deleting them would turn
+// "impossible" into "anything goes".
+func TestBoundConflictsRejectedFromStrictUpwards(t *testing.T) {
+ cases := []struct {
+ name string
+ schema string
+ }{
+ {
+ name: "minLength above maxLength",
+ schema: `{"type":"object","properties":{"a":{"type":"string","minLength":10,"maxLength":2}}}`,
+ },
+ {
+ name: "minimum above maximum",
+ schema: `{"type":"object","properties":{"a":{"type":"integer","minimum":10,"maximum":2}}}`,
+ },
+ {
+ name: "minItems above maxItems",
+ schema: `{"type":"object","properties":{"a":{"type":"array","minItems":10,"maxItems":2,"items":{"type":"string"}}}}`,
+ },
+ }
+
+ accepting := []ValidateLevel{ValidateLevelLoose, ValidateLevelLite}
+ rejecting := []ValidateLevel{ValidateLevelStrict, ValidateLevelUltra}
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ for _, level := range accepting {
+ if err := newSchemaValidator(WithValidateLevel(level)).Validate(tc.schema); err != nil {
+ t.Errorf("%s should still accept the schema, got %v", level, err)
+ }
+ }
+ for _, level := range rejecting {
+ err := newSchemaValidator(WithValidateLevel(level)).Validate(tc.schema)
+ if err == nil {
+ t.Errorf("%s should reject the schema", level)
+ continue
+ }
+ if !strings.Contains(err.Error(), "cannot be greater than") {
+ t.Errorf("%s: expected a bound conflict error, got %v", level, err)
+ }
+ }
+
+ schema, err := ParseSchema(tc.schema)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ result, _ := schema.Canonical()
+ if !strings.Contains(result, `"a":{}`) {
+ t.Errorf("expected the property to degrade to {}, got %s", result)
+ }
+ })
+ }
+}
+
+// Bounds that make sense together must survive untouched at every level.
+func TestConsistentBoundsAreUntouched(t *testing.T) {
+ const schema = `{"type":"object","properties":{"a":{"type":"string","minLength":2,"maxLength":10}}}`
+
+ for _, level := range []ValidateLevel{ValidateLevelLoose, ValidateLevelLite, ValidateLevelStrict, ValidateLevelUltra} {
+ if err := newSchemaValidator(WithValidateLevel(level)).Validate(schema); err != nil {
+ t.Errorf("%s should accept consistent bounds, got %v", level, err)
+ }
+ }
+
+ parsed, err := ParseSchema(schema)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ result, warnErr := parsed.Canonical()
+ if warnErr != nil {
+ t.Fatalf("expected no warning, got %v", warnErr)
+ }
+ for _, frag := range []string{`"minLength":2`, `"maxLength":10`} {
+ if !strings.Contains(result, frag) {
+ t.Errorf("expected %s to survive, got %s", frag, result)
+ }
+ }
+}
+
+// The node that started all of this, taken from automation_update.parameters.json.
+// The use site restates the definition's type and minLength, adds a description
+// and an unsupported format, and points at the definition with $ref. lite used
+// to answer 400 because a keyword appeared on both sides at all.
+func TestReportedRefSiblingSchemaIsAccepted(t *testing.T) {
+ const schema = `{
+ "type": "object",
+ "properties": {
+ "__schema20": {
+ "type": "string",
+ "minLength": 1,
+ "format": "uuid",
+ "description": "Target thread UUID for heartbeat automations. Prefer destination=thread for the current local thread instead of inventing or copying raw thread ids.",
+ "$ref": "#/$defs/__schema2"
+ }
+ },
+ "$defs": {
+ "__schema2": { "type": "string", "minLength": 1 }
+ }
+ }`
+
+ for _, level := range []ValidateLevel{ValidateLevelLoose, ValidateLevelLite, ValidateLevelStrict} {
+ if err := newSchemaValidator(WithValidateLevel(level)).Validate(schema); err != nil {
+ t.Errorf("%s must accept the reported node, got %v", level, err)
+ }
+ }
+
+ if err := newSchemaValidator(WithValidateLevel(ValidateLevelUltra)).Validate(schema); err == nil {
+ t.Fatal("ultra must still report the unsupported format")
+ } else if !strings.Contains(strings.ToLower(err.Error()), "unsupported keywords") {
+ t.Fatalf("expected an unsupported-keyword error, got %v", err)
+ }
+
+ parsed, err := ParseSchema(schema)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ result, warnErr := parsed.Canonical()
+ if warnErr == nil || !strings.Contains(strings.ToLower(warnErr.Error()), "format") {
+ t.Fatalf("expected a warning about format, got %v", warnErr)
+ }
+
+ for _, want := range []string{
+ `"type":"string"`,
+ `"minLength":1`,
+ `"description":"Target thread UUID for heartbeat automations. Prefer destination=thread for the current local thread instead of inventing or copying raw thread ids."`,
+ } {
+ if !strings.Contains(result, want) {
+ t.Errorf("expected %s to survive, got %s", want, result)
+ }
+ }
+ if strings.Contains(result, `"format"`) {
+ t.Errorf("format is unsupported and must be dropped, got %s", result)
+ }
+}
+
+// Constraining an instance both directly and through anyOf is a legal conjunction
+// under 2020-12, and so is requiring a property the schema does not describe or
+// listing an enum value the type rules out. None of them can be handed to the
+// enforcer as written, so only the canonicalising levels report them; lite has to
+// accept the schema and leave the rewriting to Canonical.
+func TestLiteAcceptsStructuralShapesCanonicalCanRewrite(t *testing.T) {
+ cases := []struct {
+ name string
+ in string
+ // want is the complete canonical form, so a rewrite that quietly loosens
+ // the schema cannot slip past
+ want string
+ }{
+ {
+ name: "type beside anyOf is pushed into the branches",
+ in: `{"type":"object","properties":{"v":{"type":"string","anyOf":[{"minLength":1},{"maxLength":9}]}}}`,
+ want: `{"properties":{"v":{"anyOf":[{"minLength":1,"type":"string"},{"maxLength":9,"type":"string"}]}},"type":"object"}`,
+ },
+ {
+ name: "a branch the parent type rules out is dropped",
+ in: `{"type":"object","properties":{"v":{"type":"string","anyOf":[{"minLength":1},{"type":"integer"}]}}}`,
+ want: `{"properties":{"v":{"anyOf":[{"minLength":1,"type":"string"}]}},"type":"object"}`,
+ },
+ {
+ name: "every branch ruled out leaves nothing to satisfy",
+ in: `{"type":"object","properties":{"v":{"type":"string","anyOf":[{"type":"integer"},{"type":"boolean"}]}}}`,
+ want: `{"properties":{"v":{}},"type":"object"}`,
+ },
+ {
+ name: "a contradiction behind a branch's $ref also empties the node",
+ in: `{"type":"object","properties":{"v":{"type":"string","anyOf":[{"$ref":"#/$defs/S"}]}},"$defs":{"S":{"type":"number"}}}`,
+ want: `{"$defs":{"S":{"type":"number"}},"properties":{"v":{}},"type":"object"}`,
+ },
+ {
+ name: "the stricter of the two bounds survives distribution",
+ in: `{"type":"object","properties":{"v":{"minLength":20,"anyOf":[{"type":"string","minLength":10}]}}}`,
+ want: `{"properties":{"v":{"anyOf":[{"minLength":20,"type":"string"}]}},"type":"object"}`,
+ },
+ {
+ name: "an annotation beside anyOf stays where it is",
+ in: `{"type":"object","properties":{"v":{"description":"x","anyOf":[{"type":"string"},{"type":"integer"}]}}}`,
+ want: `{"properties":{"v":{"anyOf":[{"type":"string"},{"type":"integer"}],"description":"x"}},"type":"object"}`,
+ },
+ {
+ name: "an undeclared required entry is pruned and the declared one kept",
+ in: `{"type":"object","properties":{"a":{"type":"string"}},"required":["a","b"]}`,
+ want: `{"properties":{"a":{"type":"string"}},"required":["a"],"type":"object"}`,
+ },
+ {
+ name: "an empty required entry is pruned and the declared one kept",
+ in: `{"type":"object","properties":{"a":{"type":"string"}},"required":["a",""]}`,
+ want: `{"properties":{"a":{"type":"string"}},"required":["a"],"type":"object"}`,
+ },
+ {
+ name: "required with nothing left to keep goes away",
+ in: `{"type":"object","properties":{"a":{"type":"string"}},"required":["b"]}`,
+ want: `{"properties":{"a":{"type":"string"}},"type":"object"}`,
+ },
+ {
+ name: "required without properties asserts nothing the enforcer can use",
+ in: `{"type":"object","required":["a"]}`,
+ want: `{"type":"object"}`,
+ },
+ {
+ name: "required on a non-object is a no-op and is dropped",
+ in: `{"type":"string","required":["a"]}`,
+ want: `{"type":"string"}`,
+ },
+ {
+ name: "enum values the type rules out are dropped, the rest kept",
+ in: `{"type":"object","properties":{"v":{"type":"string","enum":["a",1,"b"]}}}`,
+ want: `{"properties":{"v":{"enum":["a","b"],"type":"string"}},"type":"object"}`,
+ },
+ {
+ // Nothing satisfies the pair, and one of them has to give. The type is
+ // kept because it is the tighter of the two survivors: keeping the enum
+ // instead would accept a boolean the schema ruled out.
+ name: "an enum the type rules out entirely gives way to the type",
+ in: `{"type":"object","properties":{"v":{"type":["null"],"enum":[false]}}}`,
+ want: `{"properties":{"v":{"type":["null"]}},"type":"object"}`,
+ },
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ for _, level := range []ValidateLevel{ValidateLevelLoose, ValidateLevelLite, ValidateLevelStrict} {
+ if err := newSchemaValidator(WithValidateLevel(level)).Validate(tc.in); err != nil {
+ t.Errorf("%s must accept the schema, got %v", level, err)
+ }
+ }
+
+ parsed, err := ParseSchema(tc.in)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ got, _ := parsed.Canonical()
+ if got != tc.want {
+ t.Errorf("canonical form\n got %s\nwant %s", got, tc.want)
+ }
+ })
+ }
+}
+
+// A path naming an anyOf branch has to be followed into that branch even when it
+// is the only part of the path. Treating it as a keyword on the root instead made
+// every simplification of a root-level branch fail and throw away the whole
+// document.
+func TestSimplifyReachesARootLevelAnyOfBranch(t *testing.T) {
+ const schema = `{"anyOf":[{"$ref":"#/$defs/S","type":"string"}],"$defs":{"S":{"type":"string","maxLength":3}}}`
+
+ parsed, err := ParseSchema(schema)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+
+ // The branch restates the definition's own type, so dropping it costs nothing.
+ // What matters is that the rest of the document is still there: before the fix
+ // the failed lookup returned an empty schema.
+ got, _ := parsed.Canonical()
+ const want = `{"$defs":{"S":{"maxLength":3,"type":"string"}},"anyOf":[{"$ref":"#/$defs/S"}]}`
+ if got != want {
+ t.Errorf("canonical form\n got %s\nwant %s", got, want)
+ }
+}
+
+// The empty string is a property name like any other: 2020-12 puts no constraint
+// on the keys of properties, and the enforcer generates it, down to {"": ...}.
+// walle used to delete such a property while canonicalising, silently dropping a
+// field the caller had declared.
+func TestEmptyPropertyNameSurvives(t *testing.T) {
+ cases := []struct {
+ name string
+ in string
+ want string
+ }{
+ {
+ name: "declared as a property",
+ in: `{"type":"object","properties":{"":{"type":"string"},"a":{"type":"number"}}}`,
+ want: `{"properties":{"":{"type":"string"},"a":{"type":"number"}},"type":"object"}`,
+ },
+ {
+ name: "declared and required",
+ in: `{"type":"object","properties":{"":{"type":"string"}},"required":[""],"additionalProperties":false}`,
+ want: `{"additionalProperties":false,"properties":{"":{"type":"string"}},"required":[""],"type":"object"}`,
+ },
+ {
+ // Undeclared is the one thing that is still wrong, and it is wrong for
+ // the same reason any other undeclared name is: only that entry goes.
+ name: "required but never declared",
+ in: `{"type":"object","properties":{"a":{"type":"number"}},"required":["","a"]}`,
+ want: `{"properties":{"a":{"type":"number"}},"required":["a"],"type":"object"}`,
+ },
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ for _, level := range []ValidateLevel{ValidateLevelLite, ValidateLevelStrict} {
+ if err := newSchemaValidator(WithValidateLevel(level)).Validate(tc.in); err != nil {
+ t.Errorf("%s must accept the schema, got %v", level, err)
+ }
+ }
+
+ parsed, err := ParseSchema(tc.in)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ got, _ := parsed.Canonical()
+ if got != tc.want {
+ t.Errorf("canonical form\n got %s\nwant %s", got, tc.want)
+ }
+ })
+ }
+}
+
+// A path whose only part names an anyOf branch, such as "anyOf{0}", has to be
+// followed into that branch. Resolving it as a keyword on the root instead made
+// every simplification inside a root-level branch fail and empty the whole
+// document, which is the opposite of degrading just the field at fault.
+func TestSimplifyDegradesInsideARootLevelAnyOfBranch(t *testing.T) {
+ cases := []struct {
+ name string
+ in string
+ want string
+ }{
+ {
+ name: "an undeclared required entry inside a branch",
+ in: `{"anyOf":[{"type":"object","properties":{"a":{"type":"string"}},"required":["a","b"]}]}`,
+ want: `{"anyOf":[{"properties":{"a":{"type":"string"}},"required":["a"],"type":"object"}]}`,
+ },
+ {
+ name: "required inside a branch with nothing left to keep",
+ in: `{"anyOf":[{"type":"object","properties":{"a":{"type":"string"}},"required":["b"]}]}`,
+ want: `{"anyOf":[{"properties":{"a":{"type":"string"}},"type":"object"}]}`,
+ },
+ {
+ name: "an enum value the branch's type rules out",
+ in: `{"anyOf":[{"type":"string","enum":["a",1]}]}`,
+ want: `{"anyOf":[{"enum":["a"],"type":"string"}]}`,
+ },
+ {
+ name: "an enum the branch's type rules out entirely",
+ in: `{"anyOf":[{"type":["null"],"enum":[false]}]}`,
+ want: `{"anyOf":[{"type":["null"]}]}`,
+ },
+ {
+ name: "a nested anyOf carrying its own parent constraint",
+ in: `{"anyOf":[{"type":"string","anyOf":[{"minLength":1},{"maxLength":9}]}]}`,
+ want: `{"anyOf":[{"anyOf":[{"minLength":1,"type":"string"},{"maxLength":9,"type":"string"}]}]}`,
+ },
+ }
+
+ for _, tc := range cases {
+ t.Run(tc.name, func(t *testing.T) {
+ parsed, err := ParseSchema(tc.in)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ got, _ := parsed.Canonical()
+ if got != tc.want {
+ t.Errorf("canonical form\n got %s\nwant %s", got, tc.want)
+ }
+ })
+ }
+}
+
+// A chain of definitions where every level references the next through two
+// required properties forms a DAG, not a cycle: every definition terminates,
+// so the schema is valid. Walking each sibling's subgraph from scratch costs
+// 2^n walks -- n=30 would take hours; reusing expanded definitions brings it
+// back to milliseconds.
+func TestRefTerminationReusesExpandedDefs(t *testing.T) {
+ var sb strings.Builder
+ sb.WriteString(`{"type":"object","properties":{"root":{"$ref":"#/$defs/D1"}},"required":["root"],"$defs":{`)
+ const levels = 30
+ for i := 1; i < levels; i++ {
+ if i > 1 {
+ sb.WriteString(",")
+ }
+ fmt.Fprintf(&sb, `"D%d":{"type":"object","properties":{"p":{"$ref":"#/$defs/D%d"},"q":{"$ref":"#/$defs/D%d"}},"required":["p","q"]}`, i, i+1, i+1)
+ }
+ fmt.Fprintf(&sb, `,"D%d":{"type":"string"}}}`, levels)
+
+ schema, err := ParseSchema(sb.String())
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ if err := schema.Validate(); err != nil {
+ t.Fatalf("chain of terminating definitions should validate: %v", err)
+ }
+}
+
+// Caching a termination verdict is only sound when computing it never ran into
+// the entry stack. Here D2's verdict computed while entering through D1 is cut
+// short by the stack (D2 -> D1 -> D2), but standalone D2 terminates because
+// D1's second anyOf branch is a plain string. A poisoned cache would reject
+// this schema; it is valid: {"p":"s","q":{"y":"s"}} satisfies it.
+func TestRefTerminationVerdictIsNotCachedAcrossEntryStacks(t *testing.T) {
+ raw := `{
+ "type": "object",
+ "properties": {
+ "p": {"$ref": "#/$defs/D1"},
+ "q": {"$ref": "#/$defs/D2"}
+ },
+ "required": ["p", "q"],
+ "$defs": {
+ "D1": {
+ "anyOf": [
+ {"type": "object", "properties": {"x": {"$ref": "#/$defs/D2"}}, "required": ["x"]},
+ {"type": "string"}
+ ]
+ },
+ "D2": {"type": "object", "properties": {"y": {"$ref": "#/$defs/D1"}}, "required": ["y"]}
+ }
+ }`
+
+ schema, err := ParseSchema(raw)
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ if err := schema.Validate(); err != nil {
+ t.Fatalf("schema with a finite instance should validate: %v", err)
+ }
+}
+
+// A diamond chain that closes a cycle has no finite instance, but proving it by
+// walking costs 2^n walks: every path is cut by the cycle guard, so nothing is
+// memoizable. The shared step budget fails closed instead of hanging.
+func TestRefTerminationFailsClosedBeyondStepBudget(t *testing.T) {
+ var sb strings.Builder
+ sb.WriteString(`{"type":"object","properties":{"root":{"$ref":"#/$defs/D1"}},"required":["root"],"$defs":{`)
+ const levels = 25
+ for i := 1; i < levels; i++ {
+ if i > 1 {
+ sb.WriteString(",")
+ }
+ fmt.Fprintf(&sb, `"D%d":{"type":"object","properties":{"p":{"$ref":"#/$defs/D%d"},"q":{"$ref":"#/$defs/D%d"}},"required":["p","q"]}`, i, i+1, i+1)
+ }
+ fmt.Fprintf(&sb, `,"D%d":{"type":"object","properties":{"back":{"$ref":"#/$defs/D1"}},"required":["back"]}}}`, levels)
+
+ schema, err := ParseSchema(sb.String())
+ if err != nil {
+ t.Fatalf("failed to parse schema: %v", err)
+ }
+ err = schema.Validate()
+ if err == nil {
+ t.Fatal("expected rejection of a cycle with no finite instance")
+ }
+ if !strings.Contains(err.Error(), "too complex") && !strings.Contains(err.Error(), "infinite recursion") {
+ t.Fatalf("unexpected error: %v", err)
+ }
+}