From 9cb88c2f669c5e9574bb904fbf33652c6dc91f34 Mon Sep 17 00:00:00 2001 From: iperev Date: Sun, 20 Sep 2026 23:45:24 +0200 Subject: [PATCH] perf: bound retained indexes for rejected source branches --- .../requirementsourcecodec/codec_fuzz_test.go | 12 +- .../index_retention_test.go | 116 ++++++++++++++++++ .../requirementsourcecodec/json_index.go | 59 +++++---- internal/tools/releasechange/record_test.go | 8 +- package-lock.json | 4 +- package.json | 2 +- release/change-record.v2.json | 8 +- 7 files changed, 171 insertions(+), 38 deletions(-) create mode 100644 internal/kernel/requirementsourcecodec/index_retention_test.go diff --git a/internal/kernel/requirementsourcecodec/codec_fuzz_test.go b/internal/kernel/requirementsourcecodec/codec_fuzz_test.go index 108e54b7..cfae935d 100644 --- a/internal/kernel/requirementsourcecodec/codec_fuzz_test.go +++ b/internal/kernel/requirementsourcecodec/codec_fuzz_test.go @@ -22,12 +22,14 @@ func FuzzParseCanonicalRoundTrip(f *testing.F) { f.Add([]byte(`{}`)) f.Add([]byte(`{"schemaVersion":2,"kind":"proofkit.requirement-source"}`)) model, err := requirementsourcemodel.NormalizeWithLimits(testDraft(), modelLimits) - if err == nil { - payload, formatErr := FormatWithLimits(model, codecLimits, modelLimits) - if formatErr == nil { - f.Add(payload) - } + if err != nil { + f.Fatal(err) + } + payload, err := FormatWithLimits(model, codecLimits, modelLimits) + if err != nil { + f.Fatal(err) } + f.Add(payload) f.Fuzz(func(t *testing.T, source []byte) { assertCodecFuzzProperties(t, source, codecLimits, modelLimits) }) diff --git a/internal/kernel/requirementsourcecodec/index_retention_test.go b/internal/kernel/requirementsourcecodec/index_retention_test.go new file mode 100644 index 00000000..6efbddc8 --- /dev/null +++ b/internal/kernel/requirementsourcecodec/index_retention_test.go @@ -0,0 +1,116 @@ +package requirementsourcecodec + +import ( + "bytes" + "encoding/json" + "fmt" + "reflect" + "strconv" + "strings" + "testing" +) + +func TestRejectedContainerRetainsOnlyItsBoundary(t *testing.T) { + for _, count := range []int{16, 128, 1024} { + for _, kind := range []string{"unknown-array", "unknown-object", "wrong-array", "wrong-object"} { + t.Run(fmt.Sprintf("%s/%d", kind, count), func(t *testing.T) { + key, code := strings.Repeat("q", 4096), "unknown_field" + if strings.HasPrefix(kind, "wrong-") { + key, code = "sourceId", "invalid_type" + } + var value strings.Builder + opening, closing := "[", "]" + if strings.HasSuffix(kind, "object") { + opening, closing = "{", "}" + } + value.WriteString(opening) + for i := 0; i < count; i++ { + if i > 0 { + value.WriteString(",") + } + if opening == "{" { + value.WriteString(strconv.Quote(strconv.Itoa(i)) + ":") + } + value.WriteString("0") + } + value.WriteString(closing) + source := []byte(`{` + strconv.Quote(key) + `:` + value.String() + `}`) + expected := documentShape(compactTestModelLimits()) + indexed, err := indexJSON(source, DefaultLimits(), expected) + if err != nil { + t.Fatal(err) + } + if len(indexed.locations) != 2 { + t.Fatal("rejected descendants retained source-map paths") + } + if got := reflect.ValueOf(indexed.value.(map[string]any)[key]); got.Len() != 0 { + t.Fatal("rejected container retained child values") + } + location := indexed.locations[joinPointer("", key)] + if string(source[location.value.Start:location.value.End]) != value.String() { + t.Fatal("retained boundary span changed") + } + if ErrorCode(validateShape(indexed.value, expected, "", indexed.locations, source)) != code { + t.Fatal("shape rejection changed") + } + }) + } + } +} + +func TestDiscardedBranchesPreserveLexicalFailuresAndRecovery(t *testing.T) { + good := mustPayload(t) + baseline, err := Parse(good) + if err != nil { + t.Fatal(err) + } + want := baseline.Model.Atomic() + cases := []struct { + name, source, code, path, token string + }{ + {"unknown duplicate", `{"unknown":{"x":0,"x":1}}`, "duplicate_field", "/", `"x"`}, + {"wrong object duplicate", `{"sourceId":{"x":0,"x":1}}`, "duplicate_field", "/sourceId", `"x"`}, + {"unknown array duplicate", `{"unknown":[{"x":0,"x":1}]}`, "duplicate_field", "//0", `"x"`}, + {"wrong root duplicate", `[{"x":0,"x":1}]`, "duplicate_field", "/0", `"x"`}, + {"unknown unicode", `{"unknown":["\ud800"]}`, "invalid_unicode_escape", "//0", `"\ud800"`}, + {"wrong array unicode", `{"sourceId":["\ud800"]}`, "invalid_unicode_escape", "/sourceId/0", `"\ud800"`}, + {"lexical before shape", `{"sourceId":null,"unknown":{"x":0,"x":1}}`, "duplicate_field", "/", `"x"`}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + source := []byte(tc.source) + result, err := Parse(source) + assertDiagnostic(t, err, tc.code, tc.path) + if !reflect.DeepEqual(result, Result{}) || string(source) != tc.source { + t.Fatal("failure returned state or changed source") + } + start := int64(strings.LastIndex(tc.source, tc.token)) + if got := err.(*Error).Diagnostic().Span; got != (ByteSpan{Start: start, End: start + int64(len(tc.token))}) { + t.Fatalf("diagnostic span = %#v", got) + } + recovered, recoveryErr := Parse(good) + if recoveryErr != nil || !reflect.DeepEqual(recovered.Model.Atomic(), want) { + t.Fatal("valid recovery changed") + } + assertFuzzSourceMap(t, good, recovered.SourceMap) + }) + } + nested := []byte(`{"unknown":` + strings.Repeat("[", defaultMaxNesting) + "0" + strings.Repeat("]", defaultMaxNesting) + "}") + _, err = Parse(nested) + assertDiagnostic(t, err, "nesting_limit_exceeded", "/"+strings.Repeat("/0", defaultMaxNesting-1)) + _, err = Parse([]byte(`{"unknown":[0,]}`)) + assertDiagnostic(t, err, "invalid_syntax", "") +} + +func TestDynamicMapWrongContainerKeepsDiagnosticLocation(t *testing.T) { + source := mutateRoot(t, mustPayload(t), func(root map[string]any) { + values := root["scenarios"].([]any)[0].(map[string]any)["examples"].([]any)[0].(map[string]any)["values"].(map[string]any) + values["surface"] = json.RawMessage(`{"nested":["\ud800"]}`) + }) + _, err := Parse(source) + assertDiagnostic(t, err, "invalid_unicode_escape", "/scenarios/0/examples/0/values///0") + start := int64(bytes.Index(source, []byte(`"\ud800"`))) + if err.(*Error).Diagnostic().Span != (ByteSpan{Start: start, End: start + 8}) { + t.Fatal("nested dynamic-map scalar span changed") + } +} diff --git a/internal/kernel/requirementsourcecodec/json_index.go b/internal/kernel/requirementsourcecodec/json_index.go index 7c34dc0e..552d2744 100644 --- a/internal/kernel/requirementsourcecodec/json_index.go +++ b/internal/kernel/requirementsourcecodec/json_index.go @@ -32,7 +32,7 @@ func indexJSON(source []byte, limits Limits, expected *shape) (indexedValue, err decoder := json.NewDecoder(bytes.NewReader(source)) decoder.UseNumber() indexer := &jsonIndexer{source: source, decoder: decoder, limits: limits, locations: map[string]rawLocation{}} - value, _, err := indexer.parseValue("", "", 1, expected) + value, _, err := indexer.parseValue("", "", 1, expected, true) if err != nil { return indexedValue{}, err } @@ -48,7 +48,7 @@ func indexJSON(source []byte, limits Limits, expected *shape) (indexedValue, err return indexedValue{value: value, locations: indexer.locations}, nil } -func (indexer *jsonIndexer) parseValue(rawPath string, safePath string, depth int, expected *shape) (any, ByteSpan, error) { +func (indexer *jsonIndexer) parseValue(rawPath string, safePath string, depth int, expected *shape, retain bool) (any, ByteSpan, error) { if depth > indexer.limits.MaxNesting { offset := indexer.decoder.InputOffset() return nil, ByteSpan{}, diagnosticError(indexer.source, "nesting_limit_exceeded", safePath, ByteSpan{Start: offset, End: offset}, true) @@ -59,22 +59,26 @@ func (indexer *jsonIndexer) parseValue(rawPath string, safePath string, depth in } delimiter, isDelimiter := token.(json.Delim) if !isDelimiter { - indexer.locations[rawPath] = rawLocation{value: span} + if retain { + indexer.locations[rawPath] = rawLocation{value: span} + } return token, span, nil } switch delimiter { case '{': - return indexer.parseObject(rawPath, safePath, depth, span, expected) + return indexer.parseObject(rawPath, safePath, depth, span, expected, retain) case '[': - return indexer.parseArray(rawPath, safePath, depth, span, expected) + return indexer.parseArray(rawPath, safePath, depth, span, expected, retain) default: return nil, ByteSpan{}, diagnosticError(indexer.source, "invalid_syntax", safePath, span, true) } } -func (indexer *jsonIndexer) parseObject(rawPath string, safePath string, depth int, opening ByteSpan, expected *shape) (any, ByteSpan, error) { +func (indexer *jsonIndexer) parseObject(rawPath string, safePath string, depth int, opening ByteSpan, expected *shape, retain bool) (any, ByteSpan, error) { result := map[string]any{} seen := map[string]struct{}{} + // Keep lexical validation even when structural admission cannot visit children. + retainChildren := retain && expected != nil && expected.kind == shapeObject for indexer.decoder.More() { keyToken, keySpan, err := indexer.nextToken(safePath) if err != nil { @@ -88,17 +92,22 @@ func (indexer *jsonIndexer) parseObject(rawPath string, safePath string, depth i return nil, ByteSpan{}, diagnosticError(indexer.source, "duplicate_field", safePath, keySpan, true) } seen[key] = struct{}{} - rawChildPath := joinPointer(rawPath, key) + rawChildPath := "" + if retainChildren { + rawChildPath = joinPointer(rawPath, key) + } safeKey, childShape := safeObjectChild(expected, key) safeChildPath := joinPointer(safePath, safeKey) - value, _, err := indexer.parseValue(rawChildPath, safeChildPath, depth+1, childShape) + value, _, err := indexer.parseValue(rawChildPath, safeChildPath, depth+1, childShape, retainChildren) if err != nil { return nil, ByteSpan{}, err } - location := indexer.locations[rawChildPath] - location.key = &keySpan - indexer.locations[rawChildPath] = location - result[key] = value + if retainChildren { + location := indexer.locations[rawChildPath] + location.key = &keySpan + indexer.locations[rawChildPath] = location + result[key] = value + } } closingToken, closingSpan, err := indexer.nextToken(safePath) if err != nil { @@ -108,27 +117,33 @@ func (indexer *jsonIndexer) parseObject(rawPath string, safePath string, depth i return nil, ByteSpan{}, diagnosticError(indexer.source, "invalid_syntax", safePath, closingSpan, true) } span := ByteSpan{Start: opening.Start, End: closingSpan.End} - location := indexer.locations[rawPath] - location.value = span - indexer.locations[rawPath] = location + if retain { + indexer.locations[rawPath] = rawLocation{value: span} + } return result, span, nil } -func (indexer *jsonIndexer) parseArray(rawPath string, safePath string, depth int, opening ByteSpan, expected *shape) (any, ByteSpan, error) { +func (indexer *jsonIndexer) parseArray(rawPath string, safePath string, depth int, opening ByteSpan, expected *shape, retain bool) (any, ByteSpan, error) { result := []any{} var childShape *shape + retainChildren := retain && expected != nil && expected.kind == shapeArray if expected != nil && expected.kind == shapeArray { childShape = expected.element } for index := 0; indexer.decoder.More(); index++ { indexValue := strconv.Itoa(index) - rawChildPath := joinPointer(rawPath, indexValue) + rawChildPath := "" + if retainChildren { + rawChildPath = joinPointer(rawPath, indexValue) + } safeChildPath := joinPointer(safePath, indexValue) - value, _, err := indexer.parseValue(rawChildPath, safeChildPath, depth+1, childShape) + value, _, err := indexer.parseValue(rawChildPath, safeChildPath, depth+1, childShape, retainChildren) if err != nil { return nil, ByteSpan{}, err } - result = append(result, value) + if retainChildren { + result = append(result, value) + } } closingToken, closingSpan, err := indexer.nextToken(safePath) if err != nil { @@ -138,9 +153,9 @@ func (indexer *jsonIndexer) parseArray(rawPath string, safePath string, depth in return nil, ByteSpan{}, diagnosticError(indexer.source, "invalid_syntax", safePath, closingSpan, true) } span := ByteSpan{Start: opening.Start, End: closingSpan.End} - location := indexer.locations[rawPath] - location.value = span - indexer.locations[rawPath] = location + if retain { + indexer.locations[rawPath] = rawLocation{value: span} + } return result, span, nil } diff --git a/internal/tools/releasechange/record_test.go b/internal/tools/releasechange/record_test.go index db56fa14..a3f531ac 100644 --- a/internal/tools/releasechange/record_test.go +++ b/internal/tools/releasechange/record_test.go @@ -197,7 +197,7 @@ func TestCurrentChangeRecordNamesReviewedSemanticChanges(t *testing.T) { var currentBreakingChanges = []Change{} var currentAdditions = []Change{ - {ChangeID: "proofkit.source-model.materialization", Summary: "Remove redundant atomic and layout copies from private requirement-source normalization, size member projections from the admitted snapshot, and retain reference compaction and detached public accessors. Add allocation benchmarks and independent caller-mutation, later-normalization and retained-capacity checks. Public source formats, CLI semantics, resource limits, dependencies and supported platforms remain unchanged; no source-v2 cutover is introduced."}, + {ChangeID: "proofkit.source-codec.rejected-index-retention", Summary: "Avoid retaining descendant values and raw lexical paths beneath structurally rejected private requirement-source branches while preserving full lexical validation, duplicate detection, nesting limits, exact diagnostics and admitted source maps. Add retention, failure-isolation and recovery tests; make valid fuzz seed construction fail explicitly. Public source formats, CLI semantics, resource limits, dependencies and supported platforms remain unchanged."}, } var currentMigrationSteps = []string{} @@ -220,7 +220,7 @@ func validateCurrentChangeRecord(record Record, notes string) error { func currentExpectedReleaseNotes() string { lines := []string{ - "# @research-engineering/agentic-proofkit 0.14.20", + "# @research-engineering/agentic-proofkit 0.14.21", "", "## Breaking Contract Changes", "", @@ -273,7 +273,7 @@ func currentExpectedReleaseNotes() string { "Primary npm channel:", "", "```bash", - "npm install --save-dev --save-exact @research-engineering/agentic-proofkit@0.14.20", + "npm install --save-dev --save-exact @research-engineering/agentic-proofkit@0.14.21", "```", "", "Pre-1.0 npm consumers must keep this dependency exact-pinned.", @@ -285,7 +285,7 @@ func currentExpectedReleaseNotes() string { "## Rollback", "", "- First follow the migration and persistent-state compatibility restrictions above; changing a package pin does not roll back repository state.", - "- Pin npm consumers to the previous admitted version 0.14.19 with `npm install --save-dev --save-exact @research-engineering/agentic-proofkit@0.14.19`.", + "- Pin npm consumers to the previous admitted version 0.14.20 with `npm install --save-dev --save-exact @research-engineering/agentic-proofkit@0.14.20`.", "- Treat local package artifacts as candidates until registry identity is proven.", ) return strings.Join(lines, "\n") + "\n" diff --git a/package-lock.json b/package-lock.json index bf270880..d09d8450 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "@research-engineering/agentic-proofkit", - "version": "0.14.20", + "version": "0.14.21", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@research-engineering/agentic-proofkit", - "version": "0.14.20", + "version": "0.14.21", "cpu": [ "arm64", "x64" diff --git a/package.json b/package.json index 1bac08d4..b626d57f 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "@research-engineering/agentic-proofkit", "description": "Reusable proof profile, report, graph, and witness-planning primitives.", - "version": "0.14.20", + "version": "0.14.21", "type": "module", "license": "MIT", "sideEffects": false, diff --git a/release/change-record.v2.json b/release/change-record.v2.json index 1e3df011..6a099971 100644 --- a/release/change-record.v2.json +++ b/release/change-record.v2.json @@ -1,13 +1,13 @@ { "schemaVersion": 2, - "previousVersion": "0.14.19", - "version": "0.14.20", + "previousVersion": "0.14.20", + "version": "0.14.21", "changeClass": "compatible", "breakingChanges": [], "additions": [ { - "changeId": "proofkit.source-model.materialization", - "summary": "Remove redundant atomic and layout copies from private requirement-source normalization, size member projections from the admitted snapshot, and retain reference compaction and detached public accessors. Add allocation benchmarks and independent caller-mutation, later-normalization and retained-capacity checks. Public source formats, CLI semantics, resource limits, dependencies and supported platforms remain unchanged; no source-v2 cutover is introduced." + "changeId": "proofkit.source-codec.rejected-index-retention", + "summary": "Avoid retaining descendant values and raw lexical paths beneath structurally rejected private requirement-source branches while preserving full lexical validation, duplicate detection, nesting limits, exact diagnostics and admitted source maps. Add retention, failure-isolation and recovery tests; make valid fuzz seed construction fail explicitly. Public source formats, CLI semantics, resource limits, dependencies and supported platforms remain unchanged." } ], "migration": {